Commit 1271bb448ae for nodejs
commit 1271bb448ae46ca67ad6306b43e3aab2c5b36d3c
Author: Node.js GitHub Bot <github-bot@iojs.org>
Date: Mon Oct 5 21:40:04 2026 -0400
deps: update histogram to 0.12.0
PR-URL: https://github.com/nodejs/node/pull/66494
Reviewed-By: Colin Ihrig <cjihrig@gmail.com>
Reviewed-By: Antoine du Hamel <duhamelantoine1995@gmail.com>
Reviewed-By: James M Snell <jasnell@gmail.com>
diff --git a/deps/histogram/include/hdr/hdr_histogram.h b/deps/histogram/include/hdr/hdr_histogram.h
index 59ffe54d5cd..3dd76037ee6 100644
--- a/deps/histogram/include/hdr/hdr_histogram.h
+++ b/deps/histogram/include/hdr/hdr_histogram.h
@@ -110,6 +110,22 @@ size_t hdr_get_memory_size(struct hdr_histogram* h);
*/
bool hdr_record_value(struct hdr_histogram* h, int64_t value);
+/**
+ * Like hdr_record_value, but clamps the value into [0, highest_trackable_value]
+ * instead of rejecting out-of-range input. 0 and values below
+ * lowest_discernible_value are recorded as they are.
+ *
+ * @param h "This" pointer
+ * @param value Value to add to the histogram
+ * @return true for any value on a valid histogram.
+ */
+bool hdr_record_value_capped(struct hdr_histogram* h, int64_t value);
+
+/**
+ * Atomic version of hdr_record_value_capped, safe to call from several threads at once.
+ */
+bool hdr_record_value_capped_atomic(struct hdr_histogram* h, int64_t value);
+
/**
* Records a value in the histogram, will round this value of to a precision at or better
* than the significant_figure specified at construction time.
@@ -132,9 +148,9 @@ bool hdr_record_value_atomic(struct hdr_histogram* h, int64_t value);
*
* @param h "This" pointer
* @param value Value to add to the histogram
- * @param count Number of 'value's to add to the histogram
- * @return false if any value is larger than the highest_trackable_value and can't be recorded,
- * true otherwise.
+ * @param count Number of 'value's to add to the histogram; must be non-negative
+ * @return false if count is negative or any value is larger than the highest_trackable_value
+ * and can't be recorded, true otherwise.
*/
bool hdr_record_values(struct hdr_histogram* h, int64_t value, int64_t count);
@@ -149,9 +165,9 @@ bool hdr_record_values(struct hdr_histogram* h, int64_t value, int64_t count);
*
* @param h "This" pointer
* @param value Value to add to the histogram
- * @param count Number of 'value's to add to the histogram
- * @return false if any value is larger than the highest_trackable_value and can't be recorded,
- * true otherwise.
+ * @param count Number of 'value's to add to the histogram; must be non-negative
+ * @return false if count is negative or any value is larger than the highest_trackable_value
+ * and can't be recorded, true otherwise.
*/
bool hdr_record_values_atomic(struct hdr_histogram* h, int64_t value, int64_t count);
@@ -264,6 +280,14 @@ int64_t hdr_min(const struct hdr_histogram* h);
*/
int64_t hdr_max(const struct hdr_histogram* h);
+/**
+ * Get the total number of recorded values. Returns 0 if h is NULL. Uses an atomic
+ * load, so it can be called while other threads use the *_atomic record functions.
+ *
+ * @param h "This" pointer
+ */
+int64_t hdr_total_count(const struct hdr_histogram* h);
+
/**
* Get the value at a specific percentile.
*
@@ -433,7 +457,20 @@ void hdr_iter_linear_init(
int64_t value_units_per_bucket);
/**
- * Initialise the iterator for use with logarithmic values
+ * Change the bucket width of a linear iterator, e.g. to widen buckets once past a
+ * region of interest. The next bucket was already scheduled with the old width, so
+ * the new width applies from the bucket after it. Does nothing for an iterator that
+ * was not initialised with hdr_iter_linear_init.
+ */
+void hdr_iter_linear_set_value_units_per_bucket(struct hdr_iter* iter, int64_t value_units_per_bucket);
+
+/**
+ * Initialise the iterator for use with logarithmic values.
+ *
+ * log_base is applied as an integer step (level *= (int64_t) log_base): a
+ * fractional base truncates toward zero (2.5 behaves as 2), and any base <= 1 --
+ * including 1 < base < 2, which truncates to 1 -- terminates after the first
+ * level. Fractional bases are not supported (the Java reference iterates in double).
*/
void hdr_iter_log_init(
struct hdr_iter* iter,
@@ -505,7 +542,8 @@ int64_t hdr_median_equivalent_value(const struct hdr_histogram* h, int64_t value
/**
* Used to reset counters after importing data manually into the histogram, used by the logging code
- * and other custom serialisation tools.
+ * and other custom serialisation tools. The positive-count total saturates at
+ * INT64_MAX if it cannot be represented; the stored counts are unchanged.
*/
void hdr_reset_internal_counters(struct hdr_histogram* h);
diff --git a/deps/histogram/include/hdr/hdr_histogram_version.h b/deps/histogram/include/hdr/hdr_histogram_version.h
index 922f8fcfcda..c009dce9f20 100644
--- a/deps/histogram/include/hdr/hdr_histogram_version.h
+++ b/deps/histogram/include/hdr/hdr_histogram_version.h
@@ -7,6 +7,6 @@
#ifndef HDR_HISTOGRAM_VERSION_H
#define HDR_HISTOGRAM_VERSION_H
-#define HDR_HISTOGRAM_VERSION "0.11.10"
+#define HDR_HISTOGRAM_VERSION "0.12.0"
#endif // HDR_HISTOGRAM_VERSION_H
diff --git a/deps/histogram/src/hdr_atomic.h b/deps/histogram/src/hdr_atomic.h
index 11b0cbd3fac..cd607aef45f 100644
--- a/deps/histogram/src/hdr_atomic.h
+++ b/deps/histogram/src/hdr_atomic.h
@@ -14,29 +14,9 @@
#include <intrin.h>
#include <stdbool.h>
-static void __inline * hdr_atomic_load_pointer(void** pointer)
-{
- _ReadBarrier();
- return *pointer;
-}
-
-static void hdr_atomic_store_pointer(void** pointer, void* value)
-{
- _WriteBarrier();
- *pointer = value;
-}
-
-static int64_t __inline hdr_atomic_load_64(int64_t* field)
-{
- _ReadBarrier();
- return *field;
-}
-
-static void __inline hdr_atomic_store_64(int64_t* field, int64_t value)
-{
- _WriteBarrier();
- *field = value;
-}
+/* Type-generic so callers pass a typed T** without an incompatible (void**) cast. */
+#define hdr_atomic_load_pointer(x) (_ReadBarrier(), *(x))
+#define hdr_atomic_store_pointer(f, v) (_WriteBarrier(), (void)(*(f) = (v)))
static int64_t __inline hdr_atomic_exchange_64(volatile int64_t* field, int64_t value)
{
@@ -79,6 +59,27 @@ static bool __inline hdr_atomic_compare_exchange_64(volatile int64_t* field, int
return *expected == _InterlockedCompareExchange64(field, desired, *expected);
}
+/* A plain 64-bit access is two 32-bit accesses on 32-bit Windows and can tear. */
+static int64_t __inline hdr_atomic_load_64(int64_t* field)
+{
+#if defined(_WIN64)
+ _ReadBarrier();
+ return *field;
+#else
+ return _InterlockedCompareExchange64(field, 0, 0);
+#endif
+}
+
+static void __inline hdr_atomic_store_64(int64_t* field, int64_t value)
+{
+#if defined(_WIN64)
+ _WriteBarrier();
+ *field = value;
+#else
+ (void) hdr_atomic_exchange_64(field, value);
+#endif
+}
+
#elif defined(__ATOMIC_SEQ_CST)
#define hdr_atomic_load_pointer(x) __atomic_load_n(x, __ATOMIC_SEQ_CST)
@@ -94,17 +95,11 @@ static bool __inline hdr_atomic_compare_exchange_64(volatile int64_t* field, int
#include <stdint.h>
#include <stdbool.h>
-static inline void* hdr_atomic_load_pointer(void** pointer)
-{
- void* p = *pointer;
- asm volatile ("" ::: "memory");
- return p;
-}
-
-static inline void hdr_atomic_store_pointer(void** pointer, void* value)
-{
- asm volatile ("lock; xchgq %0, %1" : "+q" (value), "+m" (*pointer));
-}
+/* Type-generic so callers pass a typed T** without an incompatible (void**) cast. */
+#define hdr_atomic_load_pointer(x) \
+ __extension__({ __typeof__(*(x)) _p = *(x); asm volatile ("" ::: "memory"); _p; })
+#define hdr_atomic_store_pointer(f, v) \
+ __extension__({ __typeof__(*(f)) _v = (v); asm volatile ("lock; xchgq %0, %1" : "+q" (_v), "+m" (*(f))); })
static inline int64_t hdr_atomic_load_64(int64_t* field)
{
diff --git a/deps/histogram/src/hdr_histogram.c b/deps/histogram/src/hdr_histogram.c
index a34de2c1b21..3d10cf58969 100644
--- a/deps/histogram/src/hdr_histogram.c
+++ b/deps/histogram/src/hdr_histogram.c
@@ -34,9 +34,14 @@
# define HDR_UNLIKELY(x) (x)
#endif
-/* Runtime-dispatched AVX2 path: keep the rest of this TU at the project's
- baseline ISA so the shipped binary does not silently require AVX2. */
-#if (defined(__x86_64__) || defined(_M_X64) || defined(__i386__) || defined(_M_IX86)) \
+/* Runtime-dispatched AVX2 path; rest of TU stays at baseline ISA so the binary
+ doesn't silently require AVX2. 64-bit x86 + GCC/Clang only:
+ - 32-bit x86: _mm_extract_epi64 unavailable in 32-bit codegen.
+ - _MSC_VER: __builtin_cpu_supports's __cpu_model isn't linked under MSVC;
+ clang-cl also defines __x86_64__/__clang__ so this exclusion is load-bearing.
+ - __INTEL_COMPILER: ICC classic. */
+#if !defined(HDR_DISABLE_AVX2) \
+ && defined(__x86_64__) \
&& (defined(__GNUC__) || defined(__clang__)) && !defined(__INTEL_COMPILER) && !defined(_MSC_VER)
# define HDR_HAS_AVX2_DISPATCH 1
# include <immintrin.h>
@@ -187,7 +192,8 @@ static int64_t power(int64_t base, int64_t exp)
static int32_t count_leading_zeros_64(int64_t value)
{
#if defined(_MSC_VER) && !(defined(__clang__) && (defined(_M_ARM) || defined(_M_ARM64)))
- uint32_t leading_zero = 0;
+ /* _BitScanReverse writes an unsigned long */
+ unsigned long leading_zero = 0;
#if defined(_WIN64)
_BitScanReverse64(&leading_zero, value);
#else
@@ -291,12 +297,26 @@ static int64_t lowest_equivalent_value_given_bucket_indices(
int64_t hdr_next_non_equivalent_value(const struct hdr_histogram *h, int64_t value)
{
- return lowest_equivalent_value(h, value) + hdr_size_of_equivalent_value_range(h, value);
+ int64_t low = lowest_equivalent_value(h, value);
+ int64_t size = hdr_size_of_equivalent_value_range(h, value);
+ /* saturate: top-bucket low+size overflows int64 (UB) */
+ if (low > INT64_MAX - size)
+ {
+ return INT64_MAX;
+ }
+ return low + size;
}
static int64_t highest_equivalent_value(const struct hdr_histogram* h, int64_t value)
{
- return hdr_next_non_equivalent_value(h, value) - 1;
+ int64_t low = lowest_equivalent_value(h, value);
+ int64_t size = hdr_size_of_equivalent_value_range(h, value);
+ /* clamp: top-bucket low+size-1 overflows int64; keep value <= highest_equivalent_value */
+ if (low > INT64_MAX - size)
+ {
+ return INT64_MAX;
+ }
+ return low + size - 1;
}
int64_t hdr_median_equivalent_value(const struct hdr_histogram *h, int64_t value)
@@ -314,8 +334,9 @@ static int64_t non_zero_min(const struct hdr_histogram* h)
return lowest_equivalent_value(h, h->min_value);
}
-void hdr_reset_internal_counters(struct hdr_histogram* h)
+bool hdr_reset_internal_counters_checked(struct hdr_histogram* h)
{
+ bool overflow = false;
int min_non_zero_index = -1;
int max_index = -1;
int64_t observed_total_count = 0;
@@ -325,9 +346,18 @@ void hdr_reset_internal_counters(struct hdr_histogram* h)
{
int64_t count_at_index;
- if ((count_at_index = counts_get_direct(h, i)) > 0)
+ /* logical index: pair the count with hdr_value_at_index below (offset-aware) */
+ if ((count_at_index = counts_get_normalised(h, i)) > 0)
{
- observed_total_count += count_at_index;
+ if (count_at_index > INT64_MAX - observed_total_count)
+ {
+ observed_total_count = INT64_MAX;
+ overflow = true;
+ }
+ else
+ {
+ observed_total_count += count_at_index;
+ }
max_index = i;
if (min_non_zero_index == -1 && i != 0)
{
@@ -356,6 +386,12 @@ void hdr_reset_internal_counters(struct hdr_histogram* h)
}
h->total_count = observed_total_count;
+ return !overflow;
+}
+
+void hdr_reset_internal_counters(struct hdr_histogram* h)
+{
+ (void) hdr_reset_internal_counters_checked(h);
}
static int32_t buckets_needed_to_cover_value(int64_t value, int32_t sub_bucket_count, int32_t unit_magnitude)
@@ -392,9 +428,14 @@ int hdr_calculate_bucket_config(
int32_t sub_bucket_count_magnitude;
int64_t largest_value_with_single_unit_resolution;
+ /* define cfg on every reject path so a two-step-init caller that mishandles
+ the EINVAL return never reads uninitialized fields */
+ memset(cfg, 0, sizeof(*cfg));
+
if (lowest_discernible_value < 1 ||
significant_figures < 1 || 5 < significant_figures ||
- lowest_discernible_value * 2 > highest_trackable_value)
+ /* division form: lowest*2 near INT64_MAX overflows int64 (UB) */
+ lowest_discernible_value > highest_trackable_value / 2)
{
return EINVAL;
}
@@ -416,13 +457,15 @@ int hdr_calculate_bucket_config(
cfg->unit_magnitude = (int32_t) unit_magnitude;
cfg->sub_bucket_count = (int32_t) pow(2, (cfg->sub_bucket_half_count_magnitude + 1));
cfg->sub_bucket_half_count = cfg->sub_bucket_count / 2;
- cfg->sub_bucket_mask = ((int64_t) cfg->sub_bucket_count - 1) << cfg->unit_magnitude;
+ /* reject before shifting: sub_bucket_mask shift past bit 61 is signed-shift UB */
if (cfg->unit_magnitude + cfg->sub_bucket_half_count_magnitude > 61)
{
return EINVAL;
}
+ cfg->sub_bucket_mask = ((int64_t) cfg->sub_bucket_count - 1) << cfg->unit_magnitude;
+
cfg->bucket_count = buckets_needed_to_cover_value(highest_trackable_value, cfg->sub_bucket_count, (int32_t)cfg->unit_magnitude);
cfg->counts_len = (cfg->bucket_count + 1) * (cfg->sub_bucket_count / 2);
@@ -521,17 +564,9 @@ size_t hdr_get_memory_size(struct hdr_histogram *h)
/* ####### ## ######## ## ## ## ######## ###### */
-bool hdr_record_value(struct hdr_histogram* h, int64_t value)
-{
- return hdr_record_values(h, value, 1);
-}
-
-bool hdr_record_value_atomic(struct hdr_histogram* h, int64_t value)
-{
- return hdr_record_values_atomic(h, value, 1);
-}
-
-bool hdr_record_values(struct hdr_histogram* h, int64_t value, int64_t count)
+/* Shared record body. The count-sign check lives in hdr_record_values()/_atomic()
+ below, keeping the single-value hot path (count == 1, never negative) free of it. */
+static bool record_value_counted(struct hdr_histogram* h, int64_t value, int64_t count)
{
int32_t counts_index;
@@ -552,7 +587,7 @@ bool hdr_record_values(struct hdr_histogram* h, int64_t value, int64_t count)
return true;
}
-bool hdr_record_values_atomic(struct hdr_histogram* h, int64_t value, int64_t count)
+static bool record_value_counted_atomic(struct hdr_histogram* h, int64_t value, int64_t count)
{
int32_t counts_index;
@@ -562,7 +597,6 @@ bool hdr_record_values_atomic(struct hdr_histogram* h, int64_t value, int64_t co
}
counts_index = counts_index_for(h, value);
-
if ((uint32_t)counts_index >= (uint32_t)h->counts_len)
{
return false;
@@ -574,6 +608,46 @@ bool hdr_record_values_atomic(struct hdr_histogram* h, int64_t value, int64_t co
return true;
}
+bool hdr_record_value(struct hdr_histogram* h, int64_t value)
+{
+ return record_value_counted(h, value, 1);
+}
+
+bool hdr_record_value_atomic(struct hdr_histogram* h, int64_t value)
+{
+ return record_value_counted_atomic(h, value, 1);
+}
+
+bool hdr_record_value_capped(struct hdr_histogram* h, int64_t value)
+{
+ int64_t capped = (value > h->highest_trackable_value) ? h->highest_trackable_value : value;
+ return hdr_record_value(h, capped < 0 ? 0 : capped);
+}
+
+bool hdr_record_value_capped_atomic(struct hdr_histogram* h, int64_t value)
+{
+ int64_t capped = (value > h->highest_trackable_value) ? h->highest_trackable_value : value;
+ return hdr_record_value_atomic(h, capped < 0 ? 0 : capped);
+}
+
+bool hdr_record_values(struct hdr_histogram* h, int64_t value, int64_t count)
+{
+ if (count < 0) /* non-negative counts; scan assumes a monotonic prefix */
+ {
+ return false;
+ }
+ return record_value_counted(h, value, count);
+}
+
+bool hdr_record_values_atomic(struct hdr_histogram* h, int64_t value, int64_t count)
+{
+ if (count < 0) /* see hdr_record_values */
+ {
+ return false;
+ }
+ return record_value_counted_atomic(h, value, count);
+}
+
bool hdr_record_corrected_value(struct hdr_histogram* h, int64_t value, int64_t expected_interval)
{
return hdr_record_corrected_values(h, value, 1, expected_interval);
@@ -698,6 +772,12 @@ int64_t hdr_max(const struct hdr_histogram* h)
return highest_equivalent_value(h, h->max_value);
}
+int64_t hdr_total_count(const struct hdr_histogram* h)
+{
+ /* atomic load: safe to call while other threads use the *_atomic record functions */
+ return h != NULL ? hdr_atomic_load_64((int64_t*) &h->total_count) : 0;
+}
+
int64_t hdr_min(const struct hdr_histogram* h)
{
if (0 < hdr_count_at_index(h, 0))
@@ -711,12 +791,62 @@ int64_t hdr_min(const struct hdr_histogram* h)
static int64_t get_value_from_idx_up_to_count_scalar(
const struct hdr_histogram* h, int64_t count_at_percentile)
{
- int64_t count_to_idx = 0;
- for (int32_t idx = 0; idx < h->counts_len; idx++) {
- count_to_idx += h->counts[idx];
- if (count_to_idx >= count_at_percentile)
+ /* Block-summed scan: sum BLK counts, test the running total once per block,
+ and do the exact per-element walk only for the crossing block. offset != 0
+ (decoded/rotated) reads via the offset-aware accessor. */
+ enum { BLK = 4 };
+ const int64_t* counts = h->counts;
+ const int32_t n = h->counts_len;
+ int32_t idx = 0;
+ int64_t running = 0;
+
+ if (HDR_UNLIKELY(h->normalizing_index_offset != 0))
+ {
+ for (idx = 0; idx < n; idx++)
+ {
+ running += counts_get_normalised(h, idx);
+ if (running >= count_at_percentile)
+ return hdr_value_at_index(h, idx);
+ }
+ return 0;
+ }
+
+ {
+ const int32_t blk_limit = n - (n % BLK);
+ for (; idx < blk_limit; idx += BLK)
+ {
+ /* unsigned block sum: cannot overflow under valid state (matches AVX2 path) */
+ uint64_t block_sum_u = 0;
+ int32_t j;
+ for (j = 0; j < BLK; j++)
+ block_sum_u += (uint64_t)counts[idx + j];
+ if (HDR_UNLIKELY((uint64_t)running + block_sum_u >= (uint64_t)count_at_percentile))
+ {
+#if defined(__aarch64__) && defined(__clang__) && !defined(__APPLE__)
+ /* Keep crossing-block prefix sums out of the block-sum loop. */
+#pragma clang loop unroll(disable)
+#endif
+ for (j = 0; j < BLK; j++)
+ {
+ running += counts[idx + j];
+ if (running >= count_at_percentile)
+ return hdr_value_at_index(h, idx + j);
+ }
+ }
+ else
+ {
+ running += (int64_t)block_sum_u;
+ }
+ }
+ }
+
+ for (; idx < n; idx++)
+ {
+ running += counts[idx];
+ if (running >= count_at_percentile)
return hdr_value_at_index(h, idx);
}
+
return 0;
}
@@ -727,30 +857,44 @@ static int64_t get_value_from_idx_up_to_count_avx2(
{
int64_t running = 0;
int32_t idx = 0;
- const int32_t limit = h->counts_len & ~3;
-
- for (; idx < limit; idx += 4) {
- __m256i v = _mm256_loadu_si256((const __m256i*)&h->counts[idx]);
- __m128i lo = _mm256_castsi256_si128(v);
- __m128i hi = _mm256_extracti128_si256(v, 1);
+ /* 16 int64 (4x256-bit) per iteration: amortize the horizontal reduction +
+ extract + target-cross branch over 16 elements instead of 4. */
+ const int32_t limit = h->counts_len & ~15;
+
+ for (; idx < limit; idx += 16) {
+ /* prefetch 512 B ahead to hide L2/L3 latency; clamp in-bounds — a
+ past-end pointer is UB even for a hint. */
+ int32_t pf = idx + 4 * 16;
+ _mm_prefetch((const char*)&h->counts[pf < h->counts_len ? pf : h->counts_len - 1], _MM_HINT_T0);
+ __m256i a = _mm256_loadu_si256((const __m256i*)&h->counts[idx]);
+ __m256i b = _mm256_loadu_si256((const __m256i*)&h->counts[idx + 4]);
+ __m256i c = _mm256_loadu_si256((const __m256i*)&h->counts[idx + 8]);
+ __m256i d = _mm256_loadu_si256((const __m256i*)&h->counts[idx + 12]);
+ __m256i vsum = _mm256_add_epi64(_mm256_add_epi64(a, b), _mm256_add_epi64(c, d));
+ __m128i lo = _mm256_castsi256_si128(vsum);
+ __m128i hi = _mm256_extracti128_si256(vsum, 1);
__m128i s = _mm_add_epi64(lo, hi);
- /* Lanes are non-negative counts whose total fits in int64_t (total_count
- invariant), so the chunk sum cannot overflow under valid state. Use
- unsigned add to avoid signed-overflow UB if invariants are violated. */
+ /* Reduce with unsigned arithmetic to avoid signed-overflow UB. */
int64_t chunk = (int64_t)((uint64_t)_mm_extract_epi64(s, 0)
+ (uint64_t)_mm_extract_epi64(s, 1));
- if (__builtin_expect(running + chunk >= count_at_percentile, 0)) {
- for (int32_t j = idx; j < idx + 4; j++) {
- running += h->counts[j];
+ /* counts[] are non-negative (the record path rejects count < 0), so the
+ prefix sum is monotonic: block-skip is exact and only the crossing block
+ is walked. */
+ int64_t next = (int64_t)((uint64_t)running + (uint64_t)chunk);
+ if (HDR_UNLIKELY(next >= count_at_percentile)) {
+ for (int32_t j = idx; j < idx + 16; j++) {
+ running = (int64_t)((uint64_t)running + (uint64_t)h->counts[j]);
if (running >= count_at_percentile)
return hdr_value_at_index(h, j);
}
}
- running += chunk;
+ else {
+ running = next;
+ }
}
for (; idx < h->counts_len; idx++) {
- running += h->counts[idx];
+ running = (int64_t)((uint64_t)running + (uint64_t)h->counts[idx]);
if (running >= count_at_percentile)
return hdr_value_at_index(h, idx);
}
@@ -762,7 +906,8 @@ static int64_t get_value_from_idx_up_to_count(const struct hdr_histogram* h, int
{
count_at_percentile = count_at_percentile > 0 ? count_at_percentile : 1;
#ifdef HDR_HAS_AVX2_DISPATCH
- if (__builtin_cpu_supports("avx2"))
+ /* AVX2 reads counts[] directly; offset != 0 (rotated) must use the scalar scan */
+ if (h->normalizing_index_offset == 0 && __builtin_cpu_supports("avx2"))
return get_value_from_idx_up_to_count_avx2(h, count_at_percentile);
#endif
return get_value_from_idx_up_to_count_scalar(h, count_at_percentile);
@@ -801,16 +946,68 @@ int hdr_value_at_percentiles(const struct hdr_histogram *h, const double *percen
values[i] = count_at_percentile > 1 ? count_at_percentile : 1;
}
- hdr_iter_init(&iter, h);
- int64_t total = 0;
+ uint64_t total = 0; /* unsigned: no signed-overflow UB when a hostile block sum is added at once */
size_t at_pos = 0;
- while (hdr_iter_next(&iter) && at_pos < length)
+
+ if (HDR_LIKELY(h->normalizing_index_offset == 0))
{
- total += iter.count;
- while (at_pos < length && total >= values[at_pos])
+ /* Skip whole blocks that cannot reach the next target. counts[] are
+ non-negative (the record path rejects count < 0), so the prefix sum is
+ monotonic and this block-skip is exact for any valid histogram. */
+ enum { BATCH_SCAN_BLOCK = 8 };
+ const int64_t* counts = h->counts;
+ const int32_t len = h->counts_len;
+ int32_t idx = 0;
+ for (; idx + BATCH_SCAN_BLOCK <= len && at_pos < length; idx += BATCH_SCAN_BLOCK)
+ {
+ /* unsigned sum keeps the accumulation UB-free even at the int64 boundary */
+ const uint64_t s =
+ (uint64_t)counts[idx] + (uint64_t)counts[idx + 1] +
+ (uint64_t)counts[idx + 2] + (uint64_t)counts[idx + 3] +
+ (uint64_t)counts[idx + 4] + (uint64_t)counts[idx + 5] +
+ (uint64_t)counts[idx + 6] + (uint64_t)counts[idx + 7];
+ if ((int64_t)(total + s) >= values[at_pos])
+ {
+ int32_t j;
+ for (j = idx; j < idx + BATCH_SCAN_BLOCK; j++)
+ {
+ total += (uint64_t)counts[j];
+ while (at_pos < length && (int64_t)total >= values[at_pos])
+ {
+ values[at_pos] = highest_equivalent_value(h, hdr_value_at_index(h, j));
+ at_pos++;
+ }
+ }
+ }
+ else
+ {
+ total += s;
+ }
+ }
+ /* Tail: fewer than BATCH_SCAN_BLOCK counters remain. */
+ for (; idx < len && at_pos < length; idx++)
{
- values[at_pos] = highest_equivalent_value(h, iter.value);
- at_pos++;
+ total += (uint64_t)counts[idx];
+ while (at_pos < length && (int64_t)total >= values[at_pos])
+ {
+ values[at_pos] = highest_equivalent_value(h, hdr_value_at_index(h, idx));
+ at_pos++;
+ }
+ }
+ }
+ else
+ {
+ /* offset-aware fallback (normalizing_index_offset != 0): iterator
+ dereferences counts through the normalized index */
+ hdr_iter_init(&iter, h);
+ while (hdr_iter_next(&iter) && at_pos < length)
+ {
+ total += (uint64_t)iter.count;
+ while (at_pos < length && (int64_t)total >= values[at_pos])
+ {
+ values[at_pos] = highest_equivalent_value(h, iter.value);
+ at_pos++;
+ }
}
}
return 0;
@@ -819,7 +1016,8 @@ int hdr_value_at_percentiles(const struct hdr_histogram *h, const double *percen
double hdr_mean(const struct hdr_histogram* h)
{
struct hdr_iter iter;
- int64_t total = 0, count = 0;
+ double total = 0;
+ int64_t count = 0;
int64_t total_count = h->total_count;
hdr_iter_init(&iter, h);
@@ -829,11 +1027,12 @@ double hdr_mean(const struct hdr_histogram* h)
if (0 != iter.count)
{
count += iter.count;
- total += iter.count * hdr_median_equivalent_value(h, iter.value);
+ /* sum in double: count*median can overflow int64 (UB) for large values */
+ total += (double) iter.count * (double) hdr_median_equivalent_value(h, iter.value);
}
}
- return (total * 1.0) / total_count;
+ return total / total_count;
}
double hdr_stddev(const struct hdr_histogram* h)
@@ -868,11 +1067,26 @@ int64_t hdr_lowest_equivalent_value(const struct hdr_histogram* h, int64_t value
int64_t hdr_count_at_value(const struct hdr_histogram* h, int64_t value)
{
- return counts_get_normalised(h, counts_index_for(h, value));
+ int32_t counts_index;
+
+ if (value < 0) { return 0; }
+ /* value past the array's top half-bucket maps outside counts[] (OOB); count 0 */
+ counts_index = counts_index_for(h, value);
+ if ((uint32_t)counts_index >= (uint32_t)h->counts_len)
+ {
+ return 0;
+ }
+
+ return counts_get_normalised(h, counts_index);
}
int64_t hdr_count_at_index(const struct hdr_histogram* h, int32_t index)
{
+ /* reject index outside counts[] (OOB read); unsigned compare also catches negatives */
+ if ((uint32_t)index >= (uint32_t)h->counts_len)
+ {
+ return 0;
+ }
return counts_get_normalised(h, index);
}
@@ -915,7 +1129,11 @@ static bool move_next(struct hdr_iter* iter)
iter->h, bucket_index, sub_bucket_index);
iter->lowest_equivalent_value = leq;
iter->value = value;
- iter->highest_equivalent_value = leq + size_of_equivalent_value_range - 1;
+ /* saturate: top-bucket leq+size overflows int64 (UB) */
+ iter->highest_equivalent_value =
+ (leq > INT64_MAX - size_of_equivalent_value_range)
+ ? INT64_MAX
+ : leq + size_of_equivalent_value_range - 1;
iter->median_equivalent_value = leq + (size_of_equivalent_value_range >> 1);
return true;
@@ -923,7 +1141,22 @@ static bool move_next(struct hdr_iter* iter)
static int64_t peek_next_value_from_index(struct hdr_iter* iter)
{
- return hdr_value_at_index(iter->h, iter->counts_index + 1);
+ const int32_t index = iter->counts_index + 1;
+ int32_t bucket_index = (int32_t) ((uint32_t) index >> iter->h->sub_bucket_half_count_magnitude) - 1;
+ int32_t sub_bucket_index = (index & (iter->h->sub_bucket_half_count - 1)) + iter->h->sub_bucket_half_count;
+ int32_t shift;
+ if (bucket_index < 0)
+ {
+ sub_bucket_index -= iter->h->sub_bucket_half_count;
+ bucket_index = 0;
+ }
+ shift = bucket_index + iter->h->unit_magnitude;
+ /* one past the top bucket shifts into the sign bit for a near-INT64_MAX range; saturate */
+ if (shift >= 63 || (uint64_t) sub_bucket_index > ((uint64_t) INT64_MAX >> shift))
+ {
+ return INT64_MAX;
+ }
+ return value_from_index(bucket_index, sub_bucket_index, iter->h->unit_magnitude);
}
static bool next_value_greater_than_reporting_level_upper_bound(
@@ -1142,9 +1375,25 @@ static bool iter_linear_next(struct hdr_iter* iter)
{
update_iterated_values(iter, linear->next_value_reporting_level);
- linear->next_value_reporting_level += linear->value_units_per_bucket;
- linear->next_value_reporting_level_lowest_equivalent =
- lowest_equivalent_value(iter->h, linear->next_value_reporting_level);
+ /* Emit the saturated final level once before entering the terminal state. */
+ if (linear->next_value_reporting_level == INT64_MAX)
+ {
+ linear->next_value_reporting_level_lowest_equivalent = INT64_MAX;
+ }
+ else if (linear->value_units_per_bucket <= 0 ||
+ linear->next_value_reporting_level > INT64_MAX - linear->value_units_per_bucket)
+ {
+ /* step <= 0 first: never-advances (infinite loop) and guards the subtraction; second clause is the overflow guard */
+ linear->next_value_reporting_level = INT64_MAX;
+ linear->next_value_reporting_level_lowest_equivalent =
+ lowest_equivalent_value(iter->h, INT64_MAX);
+ }
+ else
+ {
+ linear->next_value_reporting_level += linear->value_units_per_bucket;
+ linear->next_value_reporting_level_lowest_equivalent =
+ lowest_equivalent_value(iter->h, linear->next_value_reporting_level);
+ }
return true;
}
@@ -1169,12 +1418,32 @@ void hdr_iter_linear_init(struct hdr_iter* iter, const struct hdr_histogram* h,
iter->specifics.linear.count_added_in_this_iteration_step = 0;
iter->specifics.linear.value_units_per_bucket = value_units_per_bucket;
- iter->specifics.linear.next_value_reporting_level = value_units_per_bucket;
- iter->specifics.linear.next_value_reporting_level_lowest_equivalent = lowest_equivalent_value(h, value_units_per_bucket);
+ if (value_units_per_bucket <= 0)
+ {
+ /* non-positive step never advances; a negative one also reaches
+ negative left-shift UB in lowest_equivalent_value below. Pin to the
+ terminating state (matches the advance-path guard). */
+ iter->specifics.linear.next_value_reporting_level = INT64_MAX;
+ iter->specifics.linear.next_value_reporting_level_lowest_equivalent = INT64_MAX;
+ }
+ else
+ {
+ iter->specifics.linear.next_value_reporting_level = value_units_per_bucket;
+ iter->specifics.linear.next_value_reporting_level_lowest_equivalent = lowest_equivalent_value(h, value_units_per_bucket);
+ }
iter->_next_fp = iter_linear_next;
}
+void hdr_iter_linear_set_value_units_per_bucket(struct hdr_iter* iter, int64_t value_units_per_bucket)
+{
+ /* specifics is a union: writing it on any other iterator would corrupt it */
+ if (iter->_next_fp == iter_linear_next)
+ {
+ iter->specifics.linear.value_units_per_bucket = value_units_per_bucket;
+ }
+}
+
/* ## ####### ###### ### ######## #### ######## ## ## ## ## #### ###### */
/* ## ## ## ## ## ## ## ## ## ## ## ## ## ### ### ## ## ## */
/* ## ## ## ## ## ## ## ## ## ## ## ## #### #### ## ## */
@@ -1199,8 +1468,27 @@ static bool log_iter_next(struct hdr_iter *iter)
{
update_iterated_values(iter, logarithmic->next_value_reporting_level);
- logarithmic->next_value_reporting_level *= (int64_t)logarithmic->log_base;
- logarithmic->next_value_reporting_level_lowest_equivalent = lowest_equivalent_value(iter->h, logarithmic->next_value_reporting_level);
+ /* Emit the saturated final level once before entering the terminal state. */
+ {
+ int64_t base = (int64_t) logarithmic->log_base;
+ if (logarithmic->next_value_reporting_level == INT64_MAX)
+ {
+ logarithmic->next_value_reporting_level_lowest_equivalent = INT64_MAX;
+ }
+ else if (base <= 1 || logarithmic->next_value_reporting_level <= 0 || logarithmic->next_value_reporting_level > INT64_MAX / base)
+ {
+ /* base <= 1 first: never-advances (infinite loop) and short-circuits /base so base==0 can't divide-by-zero; level <= 0 never advances (0*=base loops) and *=base on a negative is overflow UB; last clause is the positive-overflow guard */
+ logarithmic->next_value_reporting_level = INT64_MAX;
+ logarithmic->next_value_reporting_level_lowest_equivalent =
+ lowest_equivalent_value(iter->h, INT64_MAX);
+ }
+ else
+ {
+ logarithmic->next_value_reporting_level *= base;
+ logarithmic->next_value_reporting_level_lowest_equivalent =
+ lowest_equivalent_value(iter->h, logarithmic->next_value_reporting_level);
+ }
+ }
return true;
}
@@ -1228,7 +1516,22 @@ void hdr_iter_log_init(
iter->specifics.log.count_added_in_this_iteration_step = 0;
iter->specifics.log.log_base = log_base;
iter->specifics.log.next_value_reporting_level = value_units_first_bucket;
- iter->specifics.log.next_value_reporting_level_lowest_equivalent = lowest_equivalent_value(h, value_units_first_bucket);
+ if (value_units_first_bucket <= 0 || !isfinite(log_base) || log_base <= 1.0 ||
+ log_base >= (double) INT64_MAX)
+ {
+ /* non-positive first bucket or base <= 1 never advances; a negative
+ first bucket also reaches negative left-shift UB in
+ lowest_equivalent_value below. A non-finite (NaN/Inf) or out-of-int64-
+ range base would hit float-cast-overflow UB at the (int64_t) log_base
+ cast in log_iter_next. Pin to the terminating state (matches the
+ advance-path guard) so that cast is never reached for a bad base. */
+ iter->specifics.log.next_value_reporting_level = INT64_MAX;
+ iter->specifics.log.next_value_reporting_level_lowest_equivalent = INT64_MAX;
+ }
+ else
+ {
+ iter->specifics.log.next_value_reporting_level_lowest_equivalent = lowest_equivalent_value(h, value_units_first_bucket);
+ }
iter->_next_fp = log_iter_next;
}
diff --git a/deps/histogram/src/hdr_histogram_internal.h b/deps/histogram/src/hdr_histogram_internal.h
new file mode 100644
index 00000000000..c391d1136a7
--- /dev/null
+++ b/deps/histogram/src/hdr_histogram_internal.h
@@ -0,0 +1,24 @@
+/**
+ * hdr_histogram_internal.h
+ * Non-public helpers shared across library translation units (not installed,
+ * not part of the public API). Distinct from hdr_tests.h, which is for helpers
+ * used only by the test suite.
+ */
+#ifndef HDR_HISTOGRAM_INTERNAL_H
+#define HDR_HISTOGRAM_INTERNAL_H
+
+#include <hdr/hdr_histogram.h>
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+/* Map a recorded value to its counts[] index. Assumes value >= 0 (callers guard);
+ defined in hdr_histogram.c and used by the log codec and the packed variant. */
+int32_t counts_index_for(const struct hdr_histogram* h, int64_t value);
+
+#ifdef __cplusplus
+}
+#endif
+
+#endif
diff --git a/deps/histogram/src/hdr_tests.h b/deps/histogram/src/hdr_tests.h
index 5cc65a0c71d..28f48f5ddb3 100644
--- a/deps/histogram/src/hdr_tests.h
+++ b/deps/histogram/src/hdr_tests.h
@@ -9,7 +9,10 @@
extern "C" {
#endif
-int32_t counts_index_for(const struct hdr_histogram* h, int64_t value);
+#include "hdr_histogram_internal.h" /* counts_index_for (shared, not test-only) */
+
+/* Returns false when imported positive counts exceed INT64_MAX. */
+bool hdr_reset_internal_counters_checked(struct hdr_histogram* h);
int hdr_encode_compressed(struct hdr_histogram* h, uint8_t** compressed_histogram, size_t* compressed_len);
int hdr_decode_compressed(uint8_t* buffer, size_t length, struct hdr_histogram** histogram);
void hdr_base64_decode_block(const char* input, uint8_t* output);