From c3def0011e5b285c73b2e4461b5be9f9fabc7ffd Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Mon, 31 Aug 2026 12:16:28 -0400 Subject: [PATCH 01/36] Add Accumulator: a striped long counter primitive as an alternative to LongAdder Enum-keyed long[]-per-stripe storage with cache-line padding, threadId&mask stripe selection, and combine+reset performed atomically under each stripe's own lock -- closing the non-atomic sumThenReset() loss window LongAdder has. Includes a JMH benchmark against LongAdder and the ConcurrentHashMap.computeIfAbsent(AtomicLong::new) anti-pattern. APMLP-1779 Co-Authored-By: Claude Sonnet 5 --- .../trace/util/AccumulatorBenchmark.java | 106 +++++++++ .../java/datadog/trace/util/Accumulator.java | 205 ++++++++++++++++++ .../datadog/trace/util/AccumulatorTest.java | 170 +++++++++++++++ 3 files changed, 481 insertions(+) create mode 100644 internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java create mode 100644 internal-api/src/main/java/datadog/trace/util/Accumulator.java create mode 100644 internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java new file mode 100644 index 00000000000..d7c494ad9df --- /dev/null +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -0,0 +1,106 @@ +package datadog.trace.util; + +import static java.util.concurrent.TimeUnit.MICROSECONDS; + +import java.util.concurrent.ConcurrentHashMap; +import java.util.concurrent.atomic.AtomicLong; +import java.util.concurrent.atomic.LongAdder; +import org.openjdk.jmh.annotations.Benchmark; +import org.openjdk.jmh.annotations.BenchmarkMode; +import org.openjdk.jmh.annotations.Fork; +import org.openjdk.jmh.annotations.Measurement; +import org.openjdk.jmh.annotations.Mode; +import org.openjdk.jmh.annotations.OutputTimeUnit; +import org.openjdk.jmh.annotations.Scope; +import org.openjdk.jmh.annotations.State; +import org.openjdk.jmh.annotations.Threads; +import org.openjdk.jmh.annotations.Warmup; +import org.openjdk.jmh.infra.Blackhole; + +/** + * {@link Accumulator} vs {@link LongAdder} vs the {@code ConcurrentHashMap.computeIfAbsent(key, k + * -> new AtomicLong())} anti-pattern, at one thread (no contention) and at {@link Threads#MAX} + * (heavy contention). The CHM variant allocates its counter under the bucket's bin lock on first + * sight of a key -- exactly the pathology {@link Accumulator} exists to avoid -- so its comparison + * here is against that allocation-under-lock step, not against a pre-warmed map. + */ +@State(Scope.Benchmark) +@Warmup(iterations = 1, time = 10) +@Measurement(iterations = 3, time = 10) +@BenchmarkMode(Mode.AverageTime) +@OutputTimeUnit(MICROSECONDS) +@Fork(2) +public class AccumulatorBenchmark { + + enum Counter { + HITS + } + + private final LongAdder adder = new LongAdder(); + private final long[][] accumulator = Accumulator.create(Counter.values()); + private final ConcurrentHashMap chm = new ConcurrentHashMap<>(); + + @Benchmark + @Threads(1) + public void longAdderIncrement_lowContention() { + adder.increment(); + } + + @Benchmark + @Threads(Threads.MAX) + public void longAdderIncrement_highContention() { + adder.increment(); + } + + @Benchmark + @Threads(1) + public void accumulatorIncrement_lowContention() { + Accumulator.inc(accumulator, Counter.HITS); + } + + @Benchmark + @Threads(Threads.MAX) + public void accumulatorIncrement_highContention() { + Accumulator.inc(accumulator, Counter.HITS); + } + + @Benchmark + @Threads(1) + public void chmAtomicLongIncrement_lowContention() { + chm.computeIfAbsent("hits", k -> new AtomicLong()).incrementAndGet(); + } + + @Benchmark + @Threads(Threads.MAX) + public void chmAtomicLongIncrement_highContention() { + chm.computeIfAbsent("hits", k -> new AtomicLong()).incrementAndGet(); + } + + @Benchmark + @Threads(1) + public void longAdderSumThenReset_lowContention(Blackhole blackhole) { + adder.increment(); + blackhole.consume(adder.sumThenReset()); + } + + @Benchmark + @Threads(Threads.MAX) + public void longAdderSumThenReset_highContention(Blackhole blackhole) { + adder.increment(); + blackhole.consume(adder.sumThenReset()); + } + + @Benchmark + @Threads(1) + public void accumulatorAccumulateAnd_lowContention(Blackhole blackhole) { + Accumulator.inc(accumulator, Counter.HITS); + blackhole.consume(Accumulator.accumulateAnd(accumulator)); + } + + @Benchmark + @Threads(Threads.MAX) + public void accumulatorAccumulateAnd_highContention(Blackhole blackhole) { + Accumulator.inc(accumulator, Counter.HITS); + blackhole.consume(Accumulator.accumulateAnd(accumulator)); + } +} diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java new file mode 100644 index 00000000000..29f6d155e17 --- /dev/null +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -0,0 +1,205 @@ +package datadog.trace.util; + +import datadog.trace.api.function.Strategy; +import datadog.trace.api.function.StrategyConsumer; +import java.util.Arrays; +import java.util.function.Consumer; + +/** + * A striped accumulator primitive: {@code LongAdder}'s write scalability, without {@code + * LongAdder}'s reset hazard. + * + *

{@code LongAdder} is reached for reflexively as "a cheap atomic counter," but it solves a + * narrower problem (genuine many-thread write contention) and its {@code sumThenReset()} is + * documented as not atomic against concurrent updates: it walks its cells summing-then- + * zeroing one at a time, so an increment landing on a cell after it's summed but before it's zeroed + * is silently and permanently lost. {@link #accumulateAnd} closes that window by combining and + * resetting each stripe under the same lock that guards its writers, so no increment can land in + * the gap. + * + *

Each stripe's state is a bare {@code long[]}, not a named-field struct. An {@code enum} + * assigns a name to each position via its ordinal, so name and position are the same declaration + * and cannot drift apart. This also makes {@link #combine} and {@link #reset} generic, branchless, + * fixed-trip-count array loops -- exactly the shape C2's superword optimizer reliably + * auto-vectorizes -- so they are implemented once here instead of once per caller. + * + *

{@code
+ * enum MyCounters { FOO, BAR }
+ *
+ * long[][] data = Accumulator.create(MyCounters.values());
+ * Accumulator.inc(data, MyCounters.FOO);
+ * Accumulator.update(data, stripe -> {
+ *   Accumulator.inc(stripe, MyCounters.FOO);
+ *   Accumulator.inc(stripe, MyCounters.BAR);
+ * });
+ *
+ * long[] drained = Accumulator.accumulateAnd(data); // combine + reset, atomically per stripe
+ * long foo = drained[MyCounters.FOO.ordinal()];
+ * }
+ * + *

Non-additive counters (max, "ever seen" bitmask, first-occurrence timestamp) are out of scope: + * the per-stripe operation this class provides is {@code +=} via {@link #inc}/{@link #add}, + * combined with {@code +=} in {@link #combine}. A stripeable operator only needs to be associative + * and commutative, not literally addition, but no such escape hatch is wired up here -- add one (a + * caller-supplied {@code LongBinaryOperator} strategy) only when a real candidate needs it. + * + *

This class is a pure namespace over caller-owned {@code long[][]} state, in the same style as + * {@link Hashtable} and {@link FlatHashtable} -- it allocates no container object and is not itself + * a strategy consumer's receiver. + * + *

Not built here (deliberately): a struct-{@code T}-per-stripe fallback, for a subsystem + * whose per-stripe state doesn't fit named {@code long} slots, with mutate/combine/extract as + * {@code @Strategy}-annotated seams -- reach for it only if a real candidate can't be expressed as + * an enum-keyed {@code long[]}. Likewise a raw/embedded tier (caller owns the stripe array + * directly, no owning container) -- a reserve tool for a future {@code dd-trace-core} + * hottest-per-span-path candidate, not needed by the current reporting-cadence migration targets. + * Neither is stubbed out; build it when a real caller needs it (see APMLP-1779). + */ +public final class Accumulator { + private Accumulator() {} + + /** One full cache line of {@code long}s (64 bytes), used to pad each stripe's row. */ + private static final int CACHE_LINE_LONGS = 8; + + /** + * Creates the backing storage for an accumulator over {@code values}: one {@code long[]} row per + * stripe, sized to {@code values.length} plus at least one trailing cache line of padding so + * adjacent stripe rows don't false-share. + * + *

Stripe count is fixed at a power of two derived from {@link Runtime#availableProcessors()}; + * it is not a per-call knob (see {@link #stripeCount()}). + * + * @param values the enum constants naming each counter, e.g. {@code MyCounters.values()} + * @return a new {@code long[stripeCount][paddedWidth]} array, zero-initialized + */ + public static > long[][] create(E[] values) { + int paddedWidth = paddedWidth(values.length); + int stripes = stripeCount(); + long[][] data = new long[stripes][]; + for (int i = 0; i < stripes; i++) { + data[i] = new long[paddedWidth]; + } + return data; + } + + /** + * Increments the counter named by {@code key} in the calling thread's stripe by one. + * + *

Convenience for the common case: selects the calling thread's stripe, takes its lock, and + * increments. To perform several increments under a single held lock, use {@link #update}. + */ + public static > void inc(long[][] data, E key) { + add(data, key, 1L); + } + + /** + * Adds {@code delta} to the counter named by {@code key} in the calling thread's stripe. + * + * @see #inc(long[][], Enum) + */ + public static > void add(long[][] data, E key, long delta) { + add(stripeOf(data), key, delta); + } + + /** + * Increments the counter named by {@code key} in {@code stripe} by one, under {@code stripe}'s + * own lock. + * + *

Intended for use inside an {@link #update} lambda, where {@code stripe} is already the + * calling thread's selected row: {@code synchronized} is reentrant, so calling this here does not + * deadlock or take a second lock. + */ + public static > void inc(long[] stripe, E key) { + add(stripe, key, 1L); + } + + /** + * Adds {@code delta} to the counter named by {@code key} in {@code stripe}, under {@code + * stripe}'s own lock. + * + * @see #inc(long[], Enum) + */ + public static > void add(long[] stripe, E key, long delta) { + synchronized (stripe) { + stripe[key.ordinal()] += delta; + } + } + + /** + * Runs {@code mutator} against the calling thread's stripe under a single held lock -- the escape + * hatch for performing several related updates atomically with respect to a concurrent {@link + * #accumulateAnd}. + * + * @param mutator a strategy over the selected stripe; keep it small and non-capturing so it + * inlines into the lock's critical section + */ + @StrategyConsumer + public static void update(long[][] data, @Strategy Consumer mutator) { + long[] stripe = stripeOf(data); + synchronized (stripe) { + mutator.accept(stripe); + } + } + + /** + * Combines and resets every stripe, returning the sum. Each stripe is locked for exactly as long + * as it takes to fold its values into the result and zero it -- the same lock held by {@link + * #inc}/{@link #add}/{@link #update} -- so no writer can land an increment in the gap between + * summing and zeroing the way {@code LongAdder#sumThenReset()} allows. + * + * @return a new array the same length as one stripe's row, indexed by the enum's {@code + * ordinal()} for the positions actually in use (trailing padding positions are always zero) + */ + public static long[] accumulateAnd(long[][] data) { + long[] acc = new long[data[0].length]; + for (long[] stripe : data) { + synchronized (stripe) { + combine(acc, stripe); + reset(stripe); + } + } + return acc; + } + + /** + * {@code acc[i] += stripe[i]} for every index -- a fixed-trip-count loop C2 can auto-vectorize. + */ + private static void combine(long[] acc, long[] stripe) { + for (int i = 0; i < acc.length; i++) { + acc[i] += stripe[i]; + } + } + + /** Zeroes every position of {@code stripe}, via the JVM-intrinsic {@link Arrays#fill}. */ + private static void reset(long[] stripe) { + Arrays.fill(stripe, 0L); + } + + /** + * The calling thread's stripe: cheap masking, no allocation, no map lookup. + * + *

Multiple threads can map to the same stripe (this is masking, not a bijection); each + * stripe's own lock makes that safe, just not maximally scalable under a hash collision. + */ + private static long[] stripeOf(long[][] data) { + int mask = data.length - 1; + int idx = (int) (Thread.currentThread().getId() & mask); + return data[idx]; + } + + /** + * A fixed, power-of-two stripe count sized to {@link Runtime#availableProcessors()} (rounded down + * to the nearest power of two, minimum one). Not exposed as a per-call override: a mandatory + * sizing knob on every caller fails the "print test" of self-explanatory API design. + */ + private static int stripeCount() { + int cpus = Runtime.getRuntime().availableProcessors(); + return Integer.highestOneBit(Math.max(1, cpus)); + } + + /** Rounds {@code width} up to a whole number of cache lines, plus one full trailing line. */ + private static int paddedWidth(int width) { + int wholeLines = ((width + CACHE_LINE_LONGS - 1) / CACHE_LINE_LONGS) * CACHE_LINE_LONGS; + return wholeLines + CACHE_LINE_LONGS; + } +} diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java new file mode 100644 index 00000000000..b842de011e2 --- /dev/null +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -0,0 +1,170 @@ +package datadog.trace.util; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; + +import java.util.concurrent.CountDownLatch; +import java.util.concurrent.ExecutorService; +import java.util.concurrent.Executors; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.atomic.AtomicBoolean; +import org.junit.jupiter.api.Test; + +class AccumulatorTest { + + enum Counters { + FOO, + BAR, + BAZ + } + + @Test + void freshAccumulatorSumsToZero() { + long[][] data = Accumulator.create(Counters.values()); + long[] drained = Accumulator.accumulateAnd(data); + for (Counters c : Counters.values()) { + assertEquals(0L, drained[c.ordinal()]); + } + } + + @Test + void incIncrementsByOne() { + long[][] data = Accumulator.create(Counters.values()); + Accumulator.inc(data, Counters.FOO); + Accumulator.inc(data, Counters.FOO); + Accumulator.inc(data, Counters.BAR); + + long[] drained = Accumulator.accumulateAnd(data); + assertEquals(2L, drained[Counters.FOO.ordinal()]); + assertEquals(1L, drained[Counters.BAR.ordinal()]); + assertEquals(0L, drained[Counters.BAZ.ordinal()]); + } + + @Test + void addAppliesArbitraryDelta() { + long[][] data = Accumulator.create(Counters.values()); + Accumulator.add(data, Counters.BAZ, 41L); + Accumulator.add(data, Counters.BAZ, 1L); + + long[] drained = Accumulator.accumulateAnd(data); + assertEquals(42L, drained[Counters.BAZ.ordinal()]); + } + + @Test + void updateAppliesSeveralOpsUnderOneLock() { + long[][] data = Accumulator.create(Counters.values()); + Accumulator.update( + data, + stripe -> { + Accumulator.inc(stripe, Counters.FOO); + Accumulator.inc(stripe, Counters.FOO); + Accumulator.add(stripe, Counters.BAR, 5L); + }); + + long[] drained = Accumulator.accumulateAnd(data); + assertEquals(2L, drained[Counters.FOO.ordinal()]); + assertEquals(5L, drained[Counters.BAR.ordinal()]); + } + + @Test + void accumulateAndResetsSoASecondDrainIsZero() { + long[][] data = Accumulator.create(Counters.values()); + Accumulator.inc(data, Counters.FOO); + + long[] first = Accumulator.accumulateAnd(data); + assertEquals(1L, first[Counters.FOO.ordinal()]); + + long[] second = Accumulator.accumulateAnd(data); + for (Counters c : Counters.values()) { + assertEquals(0L, second[c.ordinal()]); + } + } + + @Test + void drainedRowsAreAllTheSameLength() { + long[][] data = Accumulator.create(Counters.values()); + long[] drained = Accumulator.accumulateAnd(data); + assertEquals(data[0].length, drained.length); + assertTrue(drained.length >= Counters.values().length); + } + + @Test + void concurrentIncrementsAreNotLost() throws InterruptedException { + long[][] data = Accumulator.create(Counters.values()); + int threadCount = 16; + int incrementsPerThread = 10_000; + + ExecutorService pool = Executors.newFixedThreadPool(threadCount); + CountDownLatch start = new CountDownLatch(1); + CountDownLatch done = new CountDownLatch(threadCount); + try { + for (int t = 0; t < threadCount; t++) { + pool.execute( + () -> { + try { + start.await(); + for (int i = 0; i < incrementsPerThread; i++) { + Accumulator.inc(data, Counters.FOO); + } + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + } finally { + done.countDown(); + } + }); + } + start.countDown(); + assertTrue(done.await(30, TimeUnit.SECONDS)); + } finally { + pool.shutdown(); + } + + long[] drained = Accumulator.accumulateAnd(data); + assertEquals((long) threadCount * incrementsPerThread, drained[Counters.FOO.ordinal()]); + } + + @Test + void concurrentAccumulateAndDuringWritesNeverExceedsWritten() throws InterruptedException { + long[][] data = Accumulator.create(Counters.values()); + int threadCount = 8; + int incrementsPerThread = 5_000; + + ExecutorService pool = Executors.newFixedThreadPool(threadCount + 1); + CountDownLatch done = new CountDownLatch(threadCount); + AtomicBoolean stop = new AtomicBoolean(false); + long[] runningTotal = {0L}; + + try { + pool.execute( + () -> { + while (!stop.get()) { + long[] drained = Accumulator.accumulateAnd(data); + synchronized (runningTotal) { + runningTotal[0] += drained[Counters.FOO.ordinal()]; + } + } + }); + + for (int t = 0; t < threadCount; t++) { + pool.execute( + () -> { + for (int i = 0; i < incrementsPerThread; i++) { + Accumulator.inc(data, Counters.FOO); + } + done.countDown(); + }); + } + + assertTrue(done.await(30, TimeUnit.SECONDS)); + stop.set(true); + long[] finalDrain = Accumulator.accumulateAnd(data); + synchronized (runningTotal) { + runningTotal[0] += finalDrain[Counters.FOO.ordinal()]; + } + + assertEquals((long) threadCount * incrementsPerThread, runningTotal[0]); + } finally { + pool.shutdown(); + } + } +} From 48e01bb6c3e508126dabb33baf298285ea4753f6 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Mon, 31 Aug 2026 13:20:43 -0400 Subject: [PATCH 02/36] Oversize Accumulator's default stripe count to reduce contention collisions Sizing stripes to exactly availableProcessors() left collisions likely under real contention (birthday-paradox: n(n-1)/(2m) expected colliding pairs), and a collision costs a blocking synchronized wait rather than LongAdder's cheap CAS retry. Doubling the stripe count (floor 4) cuts accumulatorIncrement_highContention from ~0.097 to ~0.040 us/op at the cost of a pricier but far rarer accumulateAnd drain -- the right trade since inc/add run on every call while accumulateAnd runs on a reporting cadence. Benchmark javadoc updated with the re-measured numbers. Co-Authored-By: Claude Sonnet 5 --- .../trace/util/AccumulatorBenchmark.java | 32 +++++++++++++++++++ .../java/datadog/trace/util/Accumulator.java | 19 ++++++++--- 2 files changed, 46 insertions(+), 5 deletions(-) diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java index d7c494ad9df..3a85b47afbd 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -23,6 +23,38 @@ * (heavy contention). The CHM variant allocates its counter under the bucket's bin lock on first * sight of a key -- exactly the pathology {@link Accumulator} exists to avoid -- so its comparison * here is against that allocation-under-lock step, not against a pre-warmed map. + * + *

Contention result to note: at low contention, {@code accumulatorIncrement} is + * essentially free and on par with {@code longAdderIncrement}. At {@code Threads.MAX} (10 threads + * on the measurement machine), oversizing {@link Accumulator}'s stripe count from 8 (one per core) + * to 16 (roughly 2x cores, see {@code stripeCount()}) cut {@code + * accumulatorIncrement_highContention} from ~0.097 us/op to ~0.040 us/op -- fewer threads collide + * on a stripe, so fewer of them pay {@code synchronized}'s blocking wait instead of a cheap + * fast-path lock. It is still roughly 4-5x slower than {@code longAdderIncrement} (a collision-free + * CAS retry beats even an uncontended monitor enter/exit), and {@code accumulateAnd} under + * concurrent writers got correspondingly more expensive (~7.5us to ~15.5us) since draining now + * walks twice as many stripes while writers are actively landing on them. Read {@code + * accumulatorIncrement_highContention} not as "Accumulator beats LongAdder under contention" (it + * doesn't, on this shape) but as the honest cost of the drain-under-lock design that buys atomic + * combine+reset; a caller trading that safety for raw increment throughput should measure their own + * contention level before choosing between them. + * Apple M1 Max, 10 CPUs - JDK 1.8.0_382 (Zulu) - macOS/arm64 - stripeCount() = 16 + * Benchmark Mode Cnt Score Error Units + * AccumulatorBenchmark.longAdderIncrement_lowContention avgt 6 0.007 ± 0.001 us/op + * AccumulatorBenchmark.longAdderIncrement_highContention avgt 6 0.009 ± 0.001 us/op + * AccumulatorBenchmark.accumulatorIncrement_lowContention avgt 6 0.010 ± 0.001 us/op + * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 6 0.040 ± 0.002 us/op + * AccumulatorBenchmark.chmAtomicLongIncrement_lowContention avgt 6 0.010 ± 0.001 us/op + * AccumulatorBenchmark.chmAtomicLongIncrement_highContention avgt 6 0.417 ± 0.543 us/op + * AccumulatorBenchmark.longAdderSumThenReset_lowContention avgt 6 0.012 ± 0.001 us/op + * AccumulatorBenchmark.longAdderSumThenReset_highContention avgt 6 2.433 ± 0.203 us/op + * AccumulatorBenchmark.accumulatorAccumulateAnd_lowContention avgt 6 0.162 ± 0.009 us/op + * AccumulatorBenchmark.accumulatorAccumulateAnd_highContention avgt 6 15.515 ± 4.094 us/op + * + * + *

(This run had some background noise from another session on the measurement machine; the + * {@code lowContention} rows and the {@code highContention} directional deltas are reliable, but + * treat the exact {@code highContention} magnitudes as approximate.) */ @State(Scope.Benchmark) @Warmup(iterations = 1, time = 10) diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index 29f6d155e17..3c1685c2865 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -66,8 +66,9 @@ private Accumulator() {} * stripe, sized to {@code values.length} plus at least one trailing cache line of padding so * adjacent stripe rows don't false-share. * - *

Stripe count is fixed at a power of two derived from {@link Runtime#availableProcessors()}; - * it is not a per-call knob (see {@link #stripeCount()}). + *

Stripe count is fixed at a power of two oversized to roughly 2x {@link + * Runtime#availableProcessors()} (minimum 4); it is not a per-call knob (see {@link + * #stripeCount()}). * * @param values the enum constants naming each counter, e.g. {@code MyCounters.values()} * @return a new {@code long[stripeCount][paddedWidth]} array, zero-initialized @@ -188,13 +189,21 @@ private static long[] stripeOf(long[][] data) { } /** - * A fixed, power-of-two stripe count sized to {@link Runtime#availableProcessors()} (rounded down - * to the nearest power of two, minimum one). Not exposed as a per-call override: a mandatory + * A fixed, power-of-two stripe count deliberately oversized to roughly 2x {@link + * Runtime#availableProcessors()} (minimum 4). Not exposed as a per-call override: a mandatory * sizing knob on every caller fails the "print test" of self-explanatory API design. + * + *

Sizing to exactly the core count leaves stripe collisions likely under real contention + * (birthday-paradox math: with {@code n} contending threads and {@code m} stripes, expected + * colliding pairs are {@code n(n-1)/(2m)}) -- and a collision costs a blocking {@code + * synchronized} wait, not a cheap CAS retry. Doubling the stripe count roughly quarters that + * collision count for a one-time, per-accumulator memory cost, at the price of a slightly more + * expensive (but far rarer) {@link #accumulateAnd} drain -- the right trade given {@link #inc}/ + * {@link #add} run on every call while {@link #accumulateAnd} runs on a reporting cadence. */ private static int stripeCount() { int cpus = Runtime.getRuntime().availableProcessors(); - return Integer.highestOneBit(Math.max(1, cpus)); + return Math.max(4, 2 * Integer.highestOneBit(Math.max(1, cpus))); } /** Rounds {@code width} up to a whole number of cache lines, plus one full trailing line. */ From 3629c97ae96029e5988aeebed9d7ce01f41a5202 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Mon, 31 Aug 2026 13:25:58 -0400 Subject: [PATCH 03/36] Add JOL footprint test: Accumulator vs one LongAdder per counter Accumulator's realistic alternative isn't a single LongAdder but one per counter (there's no multi-counter LongAdder). Fresh instances make LongAdder look ~15x lighter, but that's an artifact of never having grown a Cell[] table under contention. Forcing real concurrent writes shows the opposite: 4 LongAdders under contention (17,560 bytes) end up over 7x heavier than Accumulator's fixed footprint (2,384 bytes), which is paid once at creation and doesn't grow with more contention or more counters, while each contended LongAdder keeps paying independently. Co-Authored-By: Claude Sonnet 5 --- .../trace/util/AccumulatorFootprintTest.java | 147 ++++++++++++++++++ 1 file changed, 147 insertions(+) create mode 100644 internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java new file mode 100644 index 00000000000..2371419b2d8 --- /dev/null +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java @@ -0,0 +1,147 @@ +package datadog.trace.util; + +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assumptions.assumeFalse; + +import datadog.environment.JavaVirtualMachine; +import java.util.concurrent.CountDownLatch; +import java.util.concurrent.ExecutorService; +import java.util.concurrent.Executors; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.atomic.LongAdder; +import org.junit.jupiter.api.BeforeAll; +import org.junit.jupiter.api.Test; +import org.openjdk.jol.info.GraphLayout; + +/** + * Retained-footprint comparison (JOL) for {@link Accumulator} vs the alternative it actually + * displaces: one {@code LongAdder} per counter (there is no multi-counter {@code LongAdder} -- a + * caller wanting N additive counters allocates N of them, one field each, as {@code OtlpTelemetry} + * and {@code PayloadDispatcherImpl} do today). + * + *

A freshly constructed {@code LongAdder} is nearly free -- it holds no {@code Cell[]} table + * until contention forces one -- so comparing fresh instances understates its real cost and + * flatters {@code LongAdder}. {@link Accumulator} pays its full striped array up front, at + * creation, sized for {@link Runtime#availableProcessors()} regardless of whether contention ever + * materializes. The realistic comparison is therefore not fresh-vs-fresh but contended-vs-fresh: + * what each actually costs once the counters they represent are hit by real concurrent writers, as + * they are on the telemetry paths this class targets. + * + *

Measured on a 10-CPU machine (JDK 1.8.0_382 Zulu), 4 counters, {@code + * Accumulator.stripeCount()} = 16: + * + *

{@code
+ * fresh:      4 LongAdders =    160 bytes, Accumulator = 2384 bytes
+ * contended:  4 LongAdders =  17560 bytes, Accumulator = 2384 bytes
+ * }
+ * + * Finding: fresh, {@code LongAdder} looks ~15x lighter -- but that's an artifact of never having + * been written to concurrently. Once real contention forces each {@code LongAdder}'s {@code Cell[]} + * table to grow (each {@code Cell} is {@code @Contended}-padded against false sharing, the same + * problem {@link Accumulator}'s own padding solves), the four {@code LongAdder}s alone end up over + * 7x heavier than {@code Accumulator}'s entire fixed footprint -- and {@code Accumulator} does not + * grow further as more contention arrives within its existing stripe count, while every additional + * concurrently-written {@code LongAdder} keeps paying this cost independently. {@code + * Accumulator}'s up-front cost is the more predictable one: fixed at creation, independent of + * runtime contention, and shared (one striped array) across however many counters the caller's enum + * declares, rather than paid per counter. + */ +class AccumulatorFootprintTest { + + enum Counters { + REQUESTS, + ERRORS, + RETRIES, + BYTES_SENT + } + + @BeforeAll + static void assumeNotJ9Jvm() { + // JOL's GraphLayout relies on HotSpot-specific Unsafe internals and throws + // IllegalStateException on J9-based JVMs (IBM/Semeru) -- same guard as + // StringIndexFootprintTest / ScopeAndContinuationLayoutTest. + assumeFalse(JavaVirtualMachine.isJ9()); + } + + static long bytes(Object root) { + return GraphLayout.parseInstance(root).totalSize(); + } + + static LongAdder[] freshAdders() { + LongAdder[] adders = new LongAdder[Counters.values().length]; + for (int i = 0; i < adders.length; i++) { + adders[i] = new LongAdder(); + } + return adders; + } + + @Test + void freshFootprint() { + LongAdder[] adders = freshAdders(); + long[][] accumulator = Accumulator.create(Counters.values()); + + long adderBytes = bytes((Object) adders); + long accumulatorBytes = bytes(accumulator); + + System.out.printf( + "fresh: %d LongAdders = %6d bytes, Accumulator = %6d bytes%n", + adders.length, adderBytes, accumulatorBytes); + } + + /** + * Drives real multi-threaded contention against a fresh set of {@code LongAdder}s to force their + * {@code Cell[]} tables to grow, then compares against {@link Accumulator}'s fixed footprint -- + * the realistic comparison, since production callers write to these counters concurrently rather + * than leaving them untouched. + * + *

Cell-table growth is driven by JVM-internal CAS-collision detection, not something this test + * controls directly, so the exact grown size can vary by run/JVM; the one invariant asserted is + * monotonic growth (a contended footprint can only be at least the fresh one). + */ + @Test + void contendedFootprint() throws InterruptedException { + LongAdder[] adders = freshAdders(); + long freshAdderBytes = bytes((Object) adders); + + int threads = Math.max(4, Runtime.getRuntime().availableProcessors()); + ExecutorService pool = Executors.newFixedThreadPool(threads); + CountDownLatch start = new CountDownLatch(1); + CountDownLatch done = new CountDownLatch(threads); + try { + for (int t = 0; t < threads; t++) { + pool.execute( + () -> { + try { + start.await(); + long deadline = System.nanoTime() + TimeUnit.SECONDS.toNanos(2); + while (System.nanoTime() < deadline) { + for (LongAdder adder : adders) { + adder.increment(); + } + } + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + } finally { + done.countDown(); + } + }); + } + start.countDown(); + assertTrue(done.await(30, TimeUnit.SECONDS)); + } finally { + pool.shutdown(); + } + + long contendedAdderBytes = bytes((Object) adders); + long[][] accumulator = Accumulator.create(Counters.values()); + long accumulatorBytes = bytes(accumulator); + + System.out.printf( + "contended: %d LongAdders = %6d bytes, Accumulator = %6d bytes%n", + adders.length, contendedAdderBytes, accumulatorBytes); + + assertTrue( + contendedAdderBytes >= freshAdderBytes, + "contended LongAdder footprint should never shrink below the fresh footprint"); + } +} From fc2f55cd8131c064a8697036515f430bcb18ce9f Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Mon, 31 Aug 2026 14:50:51 -0400 Subject: [PATCH 04/36] Benchmark Accumulator against a per-counter-locked LongAdder alternative Tests the hypothesis that a LongAdder-based helper which actually closes the same sumThenReset() reset hazard (one LongAdder per counter, a per-counter lock guarding both increment and drain) would cost about the same as Accumulator. It doesn't -- it's a clean trade-off inversion, not a wash: Accumulator's thread-sharded stripes win ~10x on the increment path, while the per-counter design wins ~24x on drain, but only because this benchmark has a single counter (its drain cost scales with counter count; Accumulator's is fixed at stripe count). Documented as a data point, not adopted -- both designs close the hazard, and the difference is negligible next to real request/span work either way. Co-Authored-By: Claude Sonnet 5 --- .../trace/util/AccumulatorBenchmark.java | 79 +++++++++++++++++++ 1 file changed, 79 insertions(+) diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java index 3a85b47afbd..d603a78d123 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -55,6 +55,30 @@ *

(This run had some background noise from another session on the measurement machine; the * {@code lowContention} rows and the {@code highContention} directional deltas are reliable, but * treat the exact {@code highContention} magnitudes as approximate.) + * + *

{@code longAdderGroup*}: is a "just fix it with LongAdder" helper actually cheaper? + * {@code groupInc}/{@code groupAccumulateAnd} are the natural correct fix using {@code LongAdder} + * as the payload: one {@code LongAdder} per counter, with a per-counter lock guarding both + * the increment and the drain (locking only the drain does nothing -- {@code sumThenReset()}'s + * internal race is against the {@code LongAdder}'s own CAS-based {@code add()}, not against any + * lock a caller takes). This closes the same reset hazard as {@link Accumulator}, but stripes by + * counter instead of by thread. + * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 6 0.029 ± 0.051 us/op + * AccumulatorBenchmark.accumulatorAccumulateAnd_highContention avgt 6 13.431 ± 5.876 us/op + * AccumulatorBenchmark.longAdderGroupIncrement_highContention avgt 6 0.294 ± 0.088 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd_highContention avgt 6 0.549 ± 0.291 us/op + * Not "similar cost" -- a clean trade-off inversion. With this benchmark's single counter, + * {@code longAdderGroup}'s per-counter lock collapses to one lock for every thread (no thread-based + * distribution at all), so it loses badly on the write path: ~10x worse than {@code Accumulator}'s + * thread-sharded stripes. But its drain only has that one lock to acquire, so it wins big there: + * ~24x better than {@code Accumulator}, which always walks all 16 stripes on every drain regardless + * of counter count. That asymmetry is the whole story: {@code longAdderGroup}'s drain cost scales + * with number of counters (more counters -> more locks to drain), while {@code + * Accumulator}'s drain cost is fixed at stripe count, independent of counter count. Which design + * actually wins for a given caller depends on that caller's counter cardinality and whether its + * write traffic concentrates on a few hot counters (favors thread-sharding) or spreads across many + * (favors counter-sharding) -- not measured here, and worth checking against the real migration + * targets before treating either number as the general answer. */ @State(Scope.Benchmark) @Warmup(iterations = 1, time = 10) @@ -71,6 +95,35 @@ enum Counter { private final LongAdder adder = new LongAdder(); private final long[][] accumulator = Accumulator.create(Counter.values()); private final ConcurrentHashMap chm = new ConcurrentHashMap<>(); + private final LongAdder[] longAdderGroup = {new LongAdder()}; + + /** + * The natural "just use LongAdder" fix for the reset hazard: one {@code LongAdder} per counter, + * with a per-counter lock guarding both the increment and the drain -- external locking around + * only the drain does nothing, since {@code sumThenReset()}'s internal race is against the {@code + * LongAdder}'s own CAS-based {@code add()}, not against any lock a caller takes. This is the fair + * comparison point: it closes the same hazard {@link Accumulator} does, but stripes by + * counter (one lock per enum constant) instead of by thread (one lock per + * stripe, shared by all counters) -- so N threads hammering the *same* counter contend on one + * lock regardless of core count, with no thread-bucket distribution at all. + */ + private static void groupInc(LongAdder[] group, int ordinal) { + LongAdder counter = group[ordinal]; + synchronized (counter) { + counter.add(1L); + } + } + + private static long[] groupAccumulateAnd(LongAdder[] group) { + long[] acc = new long[group.length]; + for (int i = 0; i < group.length; i++) { + LongAdder counter = group[i]; + synchronized (counter) { + acc[i] = counter.sumThenReset(); + } + } + return acc; + } @Benchmark @Threads(1) @@ -135,4 +188,30 @@ public void accumulatorAccumulateAnd_highContention(Blackhole blackhole) { Accumulator.inc(accumulator, Counter.HITS); blackhole.consume(Accumulator.accumulateAnd(accumulator)); } + + @Benchmark + @Threads(1) + public void longAdderGroupIncrement_lowContention() { + groupInc(longAdderGroup, Counter.HITS.ordinal()); + } + + @Benchmark + @Threads(Threads.MAX) + public void longAdderGroupIncrement_highContention() { + groupInc(longAdderGroup, Counter.HITS.ordinal()); + } + + @Benchmark + @Threads(1) + public void longAdderGroupAccumulateAnd_lowContention(Blackhole blackhole) { + groupInc(longAdderGroup, Counter.HITS.ordinal()); + blackhole.consume(groupAccumulateAnd(longAdderGroup)); + } + + @Benchmark + @Threads(Threads.MAX) + public void longAdderGroupAccumulateAnd_highContention(Blackhole blackhole) { + groupInc(longAdderGroup, Counter.HITS.ordinal()); + blackhole.consume(groupAccumulateAnd(longAdderGroup)); + } } From 89d980c0f829c6df51d22131aa35b760c529fdb4 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Mon, 31 Aug 2026 17:01:14 -0400 Subject: [PATCH 05/36] Address review comments on Accumulator Rename accumulateAnd to accumulateAndReset, use ThreadSupport.threadId() instead of the deprecated Thread.getId(), add @GuardedBy annotations on the stripe-locked helpers, add @ParametersAreNonnullByDefault, and trim the javadoc (drop the Hashtable/FlatHashtable mention, the not-yet-built non-additive-counter escape hatch, and the C2-specific vectorization detail; shorten the LongAdder comparison). Co-Authored-By: Claude Sonnet 5 --- .../trace/util/AccumulatorBenchmark.java | 6 +-- .../java/datadog/trace/util/Accumulator.java | 53 ++++++++----------- .../datadog/trace/util/AccumulatorTest.java | 20 +++---- 3 files changed, 34 insertions(+), 45 deletions(-) diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java index d603a78d123..85fca0618b4 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -31,7 +31,7 @@ * accumulatorIncrement_highContention} from ~0.097 us/op to ~0.040 us/op -- fewer threads collide * on a stripe, so fewer of them pay {@code synchronized}'s blocking wait instead of a cheap * fast-path lock. It is still roughly 4-5x slower than {@code longAdderIncrement} (a collision-free - * CAS retry beats even an uncontended monitor enter/exit), and {@code accumulateAnd} under + * CAS retry beats even an uncontended monitor enter/exit), and {@code accumulateAndReset} under * concurrent writers got correspondingly more expensive (~7.5us to ~15.5us) since draining now * walks twice as many stripes while writers are actively landing on them. Read {@code * accumulatorIncrement_highContention} not as "Accumulator beats LongAdder under contention" (it @@ -179,14 +179,14 @@ public void longAdderSumThenReset_highContention(Blackhole blackhole) { @Threads(1) public void accumulatorAccumulateAnd_lowContention(Blackhole blackhole) { Accumulator.inc(accumulator, Counter.HITS); - blackhole.consume(Accumulator.accumulateAnd(accumulator)); + blackhole.consume(Accumulator.accumulateAndReset(accumulator)); } @Benchmark @Threads(Threads.MAX) public void accumulatorAccumulateAnd_highContention(Blackhole blackhole) { Accumulator.inc(accumulator, Counter.HITS); - blackhole.consume(Accumulator.accumulateAnd(accumulator)); + blackhole.consume(Accumulator.accumulateAndReset(accumulator)); } @Benchmark diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index 3c1685c2865..9b0eb4a6448 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -1,27 +1,27 @@ package datadog.trace.util; +import datadog.environment.ThreadSupport; import datadog.trace.api.function.Strategy; import datadog.trace.api.function.StrategyConsumer; import java.util.Arrays; import java.util.function.Consumer; +import javax.annotation.ParametersAreNonnullByDefault; +import javax.annotation.concurrent.GuardedBy; /** * A striped accumulator primitive: {@code LongAdder}'s write scalability, without {@code * LongAdder}'s reset hazard. * - *

{@code LongAdder} is reached for reflexively as "a cheap atomic counter," but it solves a - * narrower problem (genuine many-thread write contention) and its {@code sumThenReset()} is - * documented as not atomic against concurrent updates: it walks its cells summing-then- - * zeroing one at a time, so an increment landing on a cell after it's summed but before it's zeroed - * is silently and permanently lost. {@link #accumulateAnd} closes that window by combining and - * resetting each stripe under the same lock that guards its writers, so no increment can land in - * the gap. + *

{@code LongAdder#sumThenReset()} is documented as not atomic against concurrent + * updates: an increment landing on a cell after it's summed but before it's zeroed is silently and + * permanently lost. {@link #accumulateAndReset} closes that window by combining and resetting each + * stripe under the same lock that guards its writers. * *

Each stripe's state is a bare {@code long[]}, not a named-field struct. An {@code enum} * assigns a name to each position via its ordinal, so name and position are the same declaration * and cannot drift apart. This also makes {@link #combine} and {@link #reset} generic, branchless, - * fixed-trip-count array loops -- exactly the shape C2's superword optimizer reliably - * auto-vectorizes -- so they are implemented once here instead of once per caller. + * fixed-trip-count array loops -- the shape designed to take advantage of SIMD / vector operations + * on modern hardware -- so they are implemented once here instead of once per caller. * *

{@code
  * enum MyCounters { FOO, BAR }
@@ -33,28 +33,14 @@
  *   Accumulator.inc(stripe, MyCounters.BAR);
  * });
  *
- * long[] drained = Accumulator.accumulateAnd(data); // combine + reset, atomically per stripe
+ * long[] drained = Accumulator.accumulateAndReset(data); // combine + reset, atomically per stripe
  * long foo = drained[MyCounters.FOO.ordinal()];
  * }
* - *

Non-additive counters (max, "ever seen" bitmask, first-occurrence timestamp) are out of scope: - * the per-stripe operation this class provides is {@code +=} via {@link #inc}/{@link #add}, - * combined with {@code +=} in {@link #combine}. A stripeable operator only needs to be associative - * and commutative, not literally addition, but no such escape hatch is wired up here -- add one (a - * caller-supplied {@code LongBinaryOperator} strategy) only when a real candidate needs it. - * - *

This class is a pure namespace over caller-owned {@code long[][]} state, in the same style as - * {@link Hashtable} and {@link FlatHashtable} -- it allocates no container object and is not itself - * a strategy consumer's receiver. - * - *

Not built here (deliberately): a struct-{@code T}-per-stripe fallback, for a subsystem - * whose per-stripe state doesn't fit named {@code long} slots, with mutate/combine/extract as - * {@code @Strategy}-annotated seams -- reach for it only if a real candidate can't be expressed as - * an enum-keyed {@code long[]}. Likewise a raw/embedded tier (caller owns the stripe array - * directly, no owning container) -- a reserve tool for a future {@code dd-trace-core} - * hottest-per-span-path candidate, not needed by the current reporting-cadence migration targets. - * Neither is stubbed out; build it when a real caller needs it (see APMLP-1779). + *

This class is a pure namespace over caller-owned {@code long[][]} state -- it allocates no + * container object and is not itself a strategy consumer's receiver. */ +@ParametersAreNonnullByDefault public final class Accumulator { private Accumulator() {} @@ -129,7 +115,7 @@ public static > void add(long[] stripe, E key, long delta) { /** * Runs {@code mutator} against the calling thread's stripe under a single held lock -- the escape * hatch for performing several related updates atomically with respect to a concurrent {@link - * #accumulateAnd}. + * #accumulateAndReset}. * * @param mutator a strategy over the selected stripe; keep it small and non-capturing so it * inlines into the lock's critical section @@ -151,7 +137,7 @@ public static void update(long[][] data, @Strategy Consumer mutator) { * @return a new array the same length as one stripe's row, indexed by the enum's {@code * ordinal()} for the positions actually in use (trailing padding positions are always zero) */ - public static long[] accumulateAnd(long[][] data) { + public static long[] accumulateAndReset(long[][] data) { long[] acc = new long[data[0].length]; for (long[] stripe : data) { synchronized (stripe) { @@ -165,6 +151,7 @@ public static long[] accumulateAnd(long[][] data) { /** * {@code acc[i] += stripe[i]} for every index -- a fixed-trip-count loop C2 can auto-vectorize. */ + @GuardedBy("stripe") private static void combine(long[] acc, long[] stripe) { for (int i = 0; i < acc.length; i++) { acc[i] += stripe[i]; @@ -172,6 +159,7 @@ private static void combine(long[] acc, long[] stripe) { } /** Zeroes every position of {@code stripe}, via the JVM-intrinsic {@link Arrays#fill}. */ + @GuardedBy("stripe") private static void reset(long[] stripe) { Arrays.fill(stripe, 0L); } @@ -184,7 +172,7 @@ private static void reset(long[] stripe) { */ private static long[] stripeOf(long[][] data) { int mask = data.length - 1; - int idx = (int) (Thread.currentThread().getId() & mask); + int idx = (int) (ThreadSupport.threadId() & mask); return data[idx]; } @@ -198,8 +186,9 @@ private static long[] stripeOf(long[][] data) { * colliding pairs are {@code n(n-1)/(2m)}) -- and a collision costs a blocking {@code * synchronized} wait, not a cheap CAS retry. Doubling the stripe count roughly quarters that * collision count for a one-time, per-accumulator memory cost, at the price of a slightly more - * expensive (but far rarer) {@link #accumulateAnd} drain -- the right trade given {@link #inc}/ - * {@link #add} run on every call while {@link #accumulateAnd} runs on a reporting cadence. + * expensive (but far rarer) {@link #accumulateAndReset} drain -- the right trade given {@link + * #inc}/ {@link #add} run on every call while {@link #accumulateAndReset} runs on a reporting + * cadence. */ private static int stripeCount() { int cpus = Runtime.getRuntime().availableProcessors(); diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java index b842de011e2..2ddff79024f 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -21,7 +21,7 @@ enum Counters { @Test void freshAccumulatorSumsToZero() { long[][] data = Accumulator.create(Counters.values()); - long[] drained = Accumulator.accumulateAnd(data); + long[] drained = Accumulator.accumulateAndReset(data); for (Counters c : Counters.values()) { assertEquals(0L, drained[c.ordinal()]); } @@ -34,7 +34,7 @@ void incIncrementsByOne() { Accumulator.inc(data, Counters.FOO); Accumulator.inc(data, Counters.BAR); - long[] drained = Accumulator.accumulateAnd(data); + long[] drained = Accumulator.accumulateAndReset(data); assertEquals(2L, drained[Counters.FOO.ordinal()]); assertEquals(1L, drained[Counters.BAR.ordinal()]); assertEquals(0L, drained[Counters.BAZ.ordinal()]); @@ -46,7 +46,7 @@ void addAppliesArbitraryDelta() { Accumulator.add(data, Counters.BAZ, 41L); Accumulator.add(data, Counters.BAZ, 1L); - long[] drained = Accumulator.accumulateAnd(data); + long[] drained = Accumulator.accumulateAndReset(data); assertEquals(42L, drained[Counters.BAZ.ordinal()]); } @@ -61,7 +61,7 @@ void updateAppliesSeveralOpsUnderOneLock() { Accumulator.add(stripe, Counters.BAR, 5L); }); - long[] drained = Accumulator.accumulateAnd(data); + long[] drained = Accumulator.accumulateAndReset(data); assertEquals(2L, drained[Counters.FOO.ordinal()]); assertEquals(5L, drained[Counters.BAR.ordinal()]); } @@ -71,10 +71,10 @@ void accumulateAndResetsSoASecondDrainIsZero() { long[][] data = Accumulator.create(Counters.values()); Accumulator.inc(data, Counters.FOO); - long[] first = Accumulator.accumulateAnd(data); + long[] first = Accumulator.accumulateAndReset(data); assertEquals(1L, first[Counters.FOO.ordinal()]); - long[] second = Accumulator.accumulateAnd(data); + long[] second = Accumulator.accumulateAndReset(data); for (Counters c : Counters.values()) { assertEquals(0L, second[c.ordinal()]); } @@ -83,7 +83,7 @@ void accumulateAndResetsSoASecondDrainIsZero() { @Test void drainedRowsAreAllTheSameLength() { long[][] data = Accumulator.create(Counters.values()); - long[] drained = Accumulator.accumulateAnd(data); + long[] drained = Accumulator.accumulateAndReset(data); assertEquals(data[0].length, drained.length); assertTrue(drained.length >= Counters.values().length); } @@ -119,7 +119,7 @@ void concurrentIncrementsAreNotLost() throws InterruptedException { pool.shutdown(); } - long[] drained = Accumulator.accumulateAnd(data); + long[] drained = Accumulator.accumulateAndReset(data); assertEquals((long) threadCount * incrementsPerThread, drained[Counters.FOO.ordinal()]); } @@ -138,7 +138,7 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten() throws Interrupted pool.execute( () -> { while (!stop.get()) { - long[] drained = Accumulator.accumulateAnd(data); + long[] drained = Accumulator.accumulateAndReset(data); synchronized (runningTotal) { runningTotal[0] += drained[Counters.FOO.ordinal()]; } @@ -157,7 +157,7 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten() throws Interrupted assertTrue(done.await(30, TimeUnit.SECONDS)); stop.set(true); - long[] finalDrain = Accumulator.accumulateAnd(data); + long[] finalDrain = Accumulator.accumulateAndReset(data); synchronized (runningTotal) { runningTotal[0] += finalDrain[Counters.FOO.ordinal()]; } From 7107b5b5c3f25d540f5cd26b27208ca43d4c78d9 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Tue, 1 Sep 2026 13:42:44 -0400 Subject: [PATCH 06/36] Split Accumulator into a typed wrapper and a nested EmbeddingSupport The original static, allocation-free Accumulator API let a caller index its long[][] with a different enum than the one it was created for -- compiles, but silently reads/writes the wrong slot. Move that raw API into a nested EmbeddingSupport namespace, and add a top-level Accumulator that owns its storage and binds inc/add/update/ accumulateAndReset to one enum at construction, mirroring StringIndex's own EmbeddingSupport split in this package. Also fix the stripe-count javadoc: with n contending threads and m stripes, doubling m halves the expected number of colliding pairs (n(n-1)/(2m)), not quarters it. Co-Authored-By: Claude Sonnet 5 --- .../java/datadog/trace/util/Accumulator.java | 368 +++++++++++------- 1 file changed, 223 insertions(+), 145 deletions(-) diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index 9b0eb4a6448..34e8b6e84ee 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -9,195 +9,273 @@ import javax.annotation.concurrent.GuardedBy; /** - * A striped accumulator primitive: {@code LongAdder}'s write scalability, without {@code - * LongAdder}'s reset hazard. - * - *

{@code LongAdder#sumThenReset()} is documented as not atomic against concurrent - * updates: an increment landing on a cell after it's summed but before it's zeroed is silently and - * permanently lost. {@link #accumulateAndReset} closes that window by combining and resetting each - * stripe under the same lock that guards its writers. - * - *

Each stripe's state is a bare {@code long[]}, not a named-field struct. An {@code enum} - * assigns a name to each position via its ordinal, so name and position are the same declaration - * and cannot drift apart. This also makes {@link #combine} and {@link #reset} generic, branchless, - * fixed-trip-count array loops -- the shape designed to take advantage of SIMD / vector operations - * on modern hardware -- so they are implemented once here instead of once per caller. + * A typed, instance-owning wrapper over {@link EmbeddingSupport}: ties an enum's type to its + * backing {@code long[][]} at construction, so {@link #inc}/{@link #add} can't be called with a key + * from a different enum than the one this accumulator was {@link #of created} for. Costs one + * field-load indirection per call versus calling {@link EmbeddingSupport} directly -- the same + * trade {@code StringIndex} makes over its own nested {@code EmbeddingSupport}. * *

{@code
  * enum MyCounters { FOO, BAR }
  *
- * long[][] data = Accumulator.create(MyCounters.values());
- * Accumulator.inc(data, MyCounters.FOO);
- * Accumulator.update(data, stripe -> {
- *   Accumulator.inc(stripe, MyCounters.FOO);
- *   Accumulator.inc(stripe, MyCounters.BAR);
+ * Accumulator counters = Accumulator.of(MyCounters.values());
+ * counters.inc(MyCounters.FOO);
+ * counters.update(stripe -> {
+ *   Accumulator.EmbeddingSupport.inc(stripe, MyCounters.FOO);
+ *   Accumulator.EmbeddingSupport.inc(stripe, MyCounters.BAR);
  * });
  *
- * long[] drained = Accumulator.accumulateAndReset(data); // combine + reset, atomically per stripe
+ * long[] drained = counters.accumulateAndReset(); // combine + reset, atomically per stripe
  * long foo = drained[MyCounters.FOO.ordinal()];
  * }
* - *

This class is a pure namespace over caller-owned {@code long[][]} state -- it allocates no - * container object and is not itself a strategy consumer's receiver. + * @see EmbeddingSupport */ -@ParametersAreNonnullByDefault -public final class Accumulator { - private Accumulator() {} +public final class Accumulator> { + private final long[][] data; - /** One full cache line of {@code long}s (64 bytes), used to pad each stripe's row. */ - private static final int CACHE_LINE_LONGS = 8; + private Accumulator(long[][] data) { + this.data = data; + } /** - * Creates the backing storage for an accumulator over {@code values}: one {@code long[]} row per - * stripe, sized to {@code values.length} plus at least one trailing cache line of padding so - * adjacent stripe rows don't false-share. - * - *

Stripe count is fixed at a power of two oversized to roughly 2x {@link - * Runtime#availableProcessors()} (minimum 4); it is not a per-call knob (see {@link - * #stripeCount()}). - * * @param values the enum constants naming each counter, e.g. {@code MyCounters.values()} - * @return a new {@code long[stripeCount][paddedWidth]} array, zero-initialized */ - public static > long[][] create(E[] values) { - int paddedWidth = paddedWidth(values.length); - int stripes = stripeCount(); - long[][] data = new long[stripes][]; - for (int i = 0; i < stripes; i++) { - data[i] = new long[paddedWidth]; - } - return data; + public static > Accumulator of(E[] values) { + return new Accumulator<>(EmbeddingSupport.create(values)); } - /** - * Increments the counter named by {@code key} in the calling thread's stripe by one. - * - *

Convenience for the common case: selects the calling thread's stripe, takes its lock, and - * increments. To perform several increments under a single held lock, use {@link #update}. - */ - public static > void inc(long[][] data, E key) { - add(data, key, 1L); + /** Increments the counter named by {@code key} in the calling thread's stripe by one. */ + public void inc(E key) { + EmbeddingSupport.inc(data, key); } - /** - * Adds {@code delta} to the counter named by {@code key} in the calling thread's stripe. - * - * @see #inc(long[][], Enum) - */ - public static > void add(long[][] data, E key, long delta) { - add(stripeOf(data), key, delta); + /** Adds {@code delta} to the counter named by {@code key} in the calling thread's stripe. */ + public void add(E key, long delta) { + EmbeddingSupport.add(data, key, delta); } /** - * Increments the counter named by {@code key} in {@code stripe} by one, under {@code stripe}'s - * own lock. + * Runs {@code mutator} against the calling thread's stripe under a single held lock -- the escape + * hatch for performing several related updates atomically with respect to a concurrent {@link + * #accumulateAndReset}. * - *

Intended for use inside an {@link #update} lambda, where {@code stripe} is already the - * calling thread's selected row: {@code synchronized} is reentrant, so calling this here does not - * deadlock or take a second lock. + * @param mutator a strategy over the selected stripe; keep it small and non-capturing so it + * inlines into the lock's critical section */ - public static > void inc(long[] stripe, E key) { - add(stripe, key, 1L); + @StrategyConsumer + public void update(@Strategy Consumer mutator) { + EmbeddingSupport.update(data, mutator); } /** - * Adds {@code delta} to the counter named by {@code key} in {@code stripe}, under {@code - * stripe}'s own lock. + * Combines and resets every stripe, returning the sum. * - * @see #inc(long[], Enum) + * @return a new array indexed by the enum's {@code ordinal()} + * @see EmbeddingSupport#accumulateAndReset */ - public static > void add(long[] stripe, E key, long delta) { - synchronized (stripe) { - stripe[key.ordinal()] += delta; - } + public long[] accumulateAndReset() { + return EmbeddingSupport.accumulateAndReset(data); } /** - * Runs {@code mutator} against the calling thread's stripe under a single held lock -- the escape - * hatch for performing several related updates atomically with respect to a concurrent {@link - * #accumulateAndReset}. + * The static, raw-array tier of the striped accumulator primitive: {@code LongAdder}'s write + * scalability, without {@code LongAdder}'s reset hazard. * - * @param mutator a strategy over the selected stripe; keep it small and non-capturing so it - * inlines into the lock's critical section + *

{@code LongAdder#sumThenReset()} is documented as not atomic against concurrent + * updates: an increment landing on a cell after it's summed but before it's zeroed is silently + * and permanently lost. {@link #accumulateAndReset} closes that window by combining and resetting + * each stripe under the same lock that guards its writers. + * + *

Each stripe's state is a bare {@code long[]}, not a named-field struct. An {@code enum} + * assigns a name to each position via its ordinal, so name and position are the same declaration + * and cannot drift apart. This also makes {@link #combine} and {@link #reset} generic, + * branchless, fixed-trip-count array loops -- the shape designed to take advantage of SIMD / + * vector operations on modern hardware -- so they are implemented once here instead of once per + * caller. + * + *

This is a pure namespace over caller-owned {@code long[][]} state -- it allocates no + * container object and is not itself a strategy consumer's receiver. That means {@code create}'s + * type parameter is not bound to the one later {@code inc}/{@code add} calls infer: nothing stops + * a caller from indexing the same {@code long[][]} with a different enum than the one it was + * {@link #create}d for, which silently reads/writes the wrong slot rather than failing to + * compile. Prefer the owning {@link Accumulator} instance, which closes that hole for one + * field-load indirection per call; reach for this class directly only when that indirection is + * worth removing. + * + *

{@code
+   * enum MyCounters { FOO, BAR }
+   *
+   * long[][] data = Accumulator.EmbeddingSupport.create(MyCounters.values());
+   * Accumulator.EmbeddingSupport.inc(data, MyCounters.FOO);
+   * Accumulator.EmbeddingSupport.update(data, stripe -> {
+   *   Accumulator.EmbeddingSupport.inc(stripe, MyCounters.FOO);
+   *   Accumulator.EmbeddingSupport.inc(stripe, MyCounters.BAR);
+   * });
+   *
+   * long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data); // per stripe
+   * long foo = drained[MyCounters.FOO.ordinal()];
+   * }
*/ - @StrategyConsumer - public static void update(long[][] data, @Strategy Consumer mutator) { - long[] stripe = stripeOf(data); - synchronized (stripe) { - mutator.accept(stripe); + @ParametersAreNonnullByDefault + public static final class EmbeddingSupport { + private EmbeddingSupport() {} + + /** One full cache line of {@code long}s (64 bytes), used to pad each stripe's row. */ + private static final int CACHE_LINE_LONGS = 8; + + /** + * Creates the backing storage for an accumulator over {@code values}: one {@code long[]} row + * per stripe, sized to {@code values.length} plus at least one trailing cache line of padding + * so adjacent stripe rows don't false-share. + * + *

Stripe count is fixed at a power of two oversized to roughly 2x {@link + * Runtime#availableProcessors()} (minimum 4); it is not a per-call knob (see {@link + * #stripeCount()}). + * + * @param values the enum constants naming each counter, e.g. {@code MyCounters.values()} + * @return a new {@code long[stripeCount][paddedWidth]} array, zero-initialized + */ + public static > long[][] create(E[] values) { + int paddedWidth = paddedWidth(values.length); + int stripes = stripeCount(); + long[][] data = new long[stripes][]; + for (int i = 0; i < stripes; i++) { + data[i] = new long[paddedWidth]; + } + return data; } - } - /** - * Combines and resets every stripe, returning the sum. Each stripe is locked for exactly as long - * as it takes to fold its values into the result and zero it -- the same lock held by {@link - * #inc}/{@link #add}/{@link #update} -- so no writer can land an increment in the gap between - * summing and zeroing the way {@code LongAdder#sumThenReset()} allows. - * - * @return a new array the same length as one stripe's row, indexed by the enum's {@code - * ordinal()} for the positions actually in use (trailing padding positions are always zero) - */ - public static long[] accumulateAndReset(long[][] data) { - long[] acc = new long[data[0].length]; - for (long[] stripe : data) { + /** + * Increments the counter named by {@code key} in the calling thread's stripe by one. + * + *

Convenience for the common case: selects the calling thread's stripe, takes its lock, and + * increments. To perform several increments under a single held lock, use {@link #update}. + */ + public static > void inc(long[][] data, E key) { + add(data, key, 1L); + } + + /** + * Adds {@code delta} to the counter named by {@code key} in the calling thread's stripe. + * + * @see #inc(long[][], Enum) + */ + public static > void add(long[][] data, E key, long delta) { + add(stripeOf(data), key, delta); + } + + /** + * Increments the counter named by {@code key} in {@code stripe} by one, under {@code stripe}'s + * own lock. + * + *

Intended for use inside an {@link #update} lambda, where {@code stripe} is already the + * calling thread's selected row: {@code synchronized} is reentrant, so calling this here does + * not deadlock or take a second lock. + */ + public static > void inc(long[] stripe, E key) { + add(stripe, key, 1L); + } + + /** + * Adds {@code delta} to the counter named by {@code key} in {@code stripe}, under {@code + * stripe}'s own lock. + * + * @see #inc(long[], Enum) + */ + public static > void add(long[] stripe, E key, long delta) { synchronized (stripe) { - combine(acc, stripe); - reset(stripe); + stripe[key.ordinal()] += delta; } } - return acc; - } - /** - * {@code acc[i] += stripe[i]} for every index -- a fixed-trip-count loop C2 can auto-vectorize. - */ - @GuardedBy("stripe") - private static void combine(long[] acc, long[] stripe) { - for (int i = 0; i < acc.length; i++) { - acc[i] += stripe[i]; + /** + * Runs {@code mutator} against the calling thread's stripe under a single held lock -- the + * escape hatch for performing several related updates atomically with respect to a concurrent + * {@link #accumulateAndReset}. + * + * @param mutator a strategy over the selected stripe; keep it small and non-capturing so it + * inlines into the lock's critical section + */ + @StrategyConsumer + public static void update(long[][] data, @Strategy Consumer mutator) { + long[] stripe = stripeOf(data); + synchronized (stripe) { + mutator.accept(stripe); + } } - } - /** Zeroes every position of {@code stripe}, via the JVM-intrinsic {@link Arrays#fill}. */ - @GuardedBy("stripe") - private static void reset(long[] stripe) { - Arrays.fill(stripe, 0L); - } + /** + * Combines and resets every stripe, returning the sum. Each stripe is locked for exactly as + * long as it takes to fold its values into the result and zero it -- the same lock held by + * {@link #inc}/{@link #add}/{@link #update} -- so no writer can land an increment in the gap + * between summing and zeroing the way {@code LongAdder#sumThenReset()} allows. + * + * @return a new array the same length as one stripe's row, indexed by the enum's {@code + * ordinal()} for the positions actually in use (trailing padding positions are always zero) + */ + public static long[] accumulateAndReset(long[][] data) { + long[] acc = new long[data[0].length]; + for (long[] stripe : data) { + synchronized (stripe) { + combine(acc, stripe); + reset(stripe); + } + } + return acc; + } - /** - * The calling thread's stripe: cheap masking, no allocation, no map lookup. - * - *

Multiple threads can map to the same stripe (this is masking, not a bijection); each - * stripe's own lock makes that safe, just not maximally scalable under a hash collision. - */ - private static long[] stripeOf(long[][] data) { - int mask = data.length - 1; - int idx = (int) (ThreadSupport.threadId() & mask); - return data[idx]; - } + /** + * {@code acc[i] += stripe[i]} for every index -- a fixed-trip-count loop C2 can auto-vectorize. + */ + @GuardedBy("stripe") + private static void combine(long[] acc, long[] stripe) { + for (int i = 0; i < acc.length; i++) { + acc[i] += stripe[i]; + } + } - /** - * A fixed, power-of-two stripe count deliberately oversized to roughly 2x {@link - * Runtime#availableProcessors()} (minimum 4). Not exposed as a per-call override: a mandatory - * sizing knob on every caller fails the "print test" of self-explanatory API design. - * - *

Sizing to exactly the core count leaves stripe collisions likely under real contention - * (birthday-paradox math: with {@code n} contending threads and {@code m} stripes, expected - * colliding pairs are {@code n(n-1)/(2m)}) -- and a collision costs a blocking {@code - * synchronized} wait, not a cheap CAS retry. Doubling the stripe count roughly quarters that - * collision count for a one-time, per-accumulator memory cost, at the price of a slightly more - * expensive (but far rarer) {@link #accumulateAndReset} drain -- the right trade given {@link - * #inc}/ {@link #add} run on every call while {@link #accumulateAndReset} runs on a reporting - * cadence. - */ - private static int stripeCount() { - int cpus = Runtime.getRuntime().availableProcessors(); - return Math.max(4, 2 * Integer.highestOneBit(Math.max(1, cpus))); - } + /** Zeroes every position of {@code stripe}, via the JVM-intrinsic {@link Arrays#fill}. */ + @GuardedBy("stripe") + private static void reset(long[] stripe) { + Arrays.fill(stripe, 0L); + } + + /** + * The calling thread's stripe: cheap masking, no allocation, no map lookup. + * + *

Multiple threads can map to the same stripe (this is masking, not a bijection); each + * stripe's own lock makes that safe, just not maximally scalable under a hash collision. + */ + private static long[] stripeOf(long[][] data) { + int mask = data.length - 1; + int idx = (int) (ThreadSupport.threadId() & mask); + return data[idx]; + } + + /** + * A fixed, power-of-two stripe count deliberately oversized to roughly 2x {@link + * Runtime#availableProcessors()} (minimum 4). Not exposed as a per-call override: a mandatory + * sizing knob on every caller fails the "print test" of self-explanatory API design. + * + *

Sizing to exactly the core count leaves stripe collisions likely under real contention + * (birthday-paradox math: with {@code n} contending threads and {@code m} stripes, expected + * colliding pairs are {@code n(n-1)/(2m)}) -- and a collision costs a blocking {@code + * synchronized} wait, not a cheap CAS retry. Doubling the stripe count roughly halves that + * collision count for a one-time, per-accumulator memory cost, at the price of a slightly more + * expensive (but far rarer) {@link #accumulateAndReset} drain -- the right trade given {@link + * #inc}/ {@link #add} run on every call while {@link #accumulateAndReset} runs on a reporting + * cadence. + */ + private static int stripeCount() { + int cpus = Runtime.getRuntime().availableProcessors(); + return Math.max(4, 2 * Integer.highestOneBit(Math.max(1, cpus))); + } - /** Rounds {@code width} up to a whole number of cache lines, plus one full trailing line. */ - private static int paddedWidth(int width) { - int wholeLines = ((width + CACHE_LINE_LONGS - 1) / CACHE_LINE_LONGS) * CACHE_LINE_LONGS; - return wholeLines + CACHE_LINE_LONGS; + /** Rounds {@code width} up to a whole number of cache lines, plus one full trailing line. */ + private static int paddedWidth(int width) { + int wholeLines = ((width + CACHE_LINE_LONGS - 1) / CACHE_LINE_LONGS) * CACHE_LINE_LONGS; + return wholeLines + CACHE_LINE_LONGS; + } } } From 175154ed5e4d30bfc4b8fc3ccb3874047350b721 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Tue, 1 Sep 2026 13:42:55 -0400 Subject: [PATCH 07/36] Update Accumulator tests/benchmark for the EmbeddingSupport split - Update call sites to Accumulator.EmbeddingSupport.*, and add a test covering the new typed Accumulator wrapper. - Fix a race in concurrentAccumulateAndDuringWritesNeverExceedsWritten: join the background drainer via Future.get() before the final drain and assertion, instead of racing it. - Assert accumulatorBytes < contendedAdderBytes in contendedFootprint -- the actual claim the test exists to back up, not just that the LongAdder side didn't shrink. - Correct the CHM benchmark's javadoc: the benchmark-scoped map means only the first warmup invocation allocates under the bin lock; every sampled op hits an already-warmed computeIfAbsent lookup. - Add a @Group-based accumulatorMixed-write/accumulatorMixed-drain pair modeling "many writers, one rare drainer," alongside the existing @Threads(MAX) benchmark kept as a documented worst-case upper bound. Co-Authored-By: Claude Sonnet 5 --- .../trace/util/AccumulatorBenchmark.java | 59 ++++++++-- .../trace/util/AccumulatorFootprintTest.java | 9 +- .../datadog/trace/util/AccumulatorTest.java | 102 +++++++++++------- 3 files changed, 118 insertions(+), 52 deletions(-) diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java index 85fca0618b4..5b6299a7eae 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -8,6 +8,8 @@ import org.openjdk.jmh.annotations.Benchmark; import org.openjdk.jmh.annotations.BenchmarkMode; import org.openjdk.jmh.annotations.Fork; +import org.openjdk.jmh.annotations.Group; +import org.openjdk.jmh.annotations.GroupThreads; import org.openjdk.jmh.annotations.Measurement; import org.openjdk.jmh.annotations.Mode; import org.openjdk.jmh.annotations.OutputTimeUnit; @@ -20,9 +22,15 @@ /** * {@link Accumulator} vs {@link LongAdder} vs the {@code ConcurrentHashMap.computeIfAbsent(key, k * -> new AtomicLong())} anti-pattern, at one thread (no contention) and at {@link Threads#MAX} - * (heavy contention). The CHM variant allocates its counter under the bucket's bin lock on first - * sight of a key -- exactly the pathology {@link Accumulator} exists to avoid -- so its comparison - * here is against that allocation-under-lock step, not against a pre-warmed map. + * (heavy contention). The CHM variant allocates its counter under the bucket's bin lock the first + * time its one constant key is seen -- exactly the pathology {@link Accumulator} exists to avoid -- + * but since the map is a {@code @State(Scope.Benchmark)} field shared across the whole run, that + * allocation happens exactly once; every sampled op after it hits the warmed, already-present fast + * path. So this measures steady-state {@code computeIfAbsent} lookup overhead on an + * already-populated map, not the one-time allocation-under-lock cost -- still a useful number (a + * fixed, small key set that's allocated once and hit for the life of the process, as {@code + * WafMetricCollector}-style CHM counters are, spends nearly all its time in this same warmed path), + * just not the pathology the name of this benchmark might suggest. * *

Contention result to note: at low contention, {@code accumulatorIncrement} is * essentially free and on par with {@code longAdderIncrement}. At {@code Threads.MAX} (10 threads @@ -93,7 +101,7 @@ enum Counter { } private final LongAdder adder = new LongAdder(); - private final long[][] accumulator = Accumulator.create(Counter.values()); + private final long[][] accumulator = Accumulator.EmbeddingSupport.create(Counter.values()); private final ConcurrentHashMap chm = new ConcurrentHashMap<>(); private final LongAdder[] longAdderGroup = {new LongAdder()}; @@ -140,13 +148,13 @@ public void longAdderIncrement_highContention() { @Benchmark @Threads(1) public void accumulatorIncrement_lowContention() { - Accumulator.inc(accumulator, Counter.HITS); + Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); } @Benchmark @Threads(Threads.MAX) public void accumulatorIncrement_highContention() { - Accumulator.inc(accumulator, Counter.HITS); + Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); } @Benchmark @@ -178,15 +186,46 @@ public void longAdderSumThenReset_highContention(Blackhole blackhole) { @Benchmark @Threads(1) public void accumulatorAccumulateAnd_lowContention(Blackhole blackhole) { - Accumulator.inc(accumulator, Counter.HITS); - blackhole.consume(Accumulator.accumulateAndReset(accumulator)); + Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); + blackhole.consume(Accumulator.EmbeddingSupport.accumulateAndReset(accumulator)); } + /** + * A deliberately pessimistic topology: every thread both writes and drains on every op, so {@code + * Threads.MAX} threads are all draining concurrently. Real callers don't do this -- see {@code + * accumulatorMixed-write}/{@code accumulatorMixed-drain} below for the "many writers, one rare + * drainer" shape this class actually targets. Kept as the worst-case upper bound: no production + * topology should be more contended on {@link Accumulator.EmbeddingSupport#accumulateAndReset} + * than this. + */ @Benchmark @Threads(Threads.MAX) public void accumulatorAccumulateAnd_highContention(Blackhole blackhole) { - Accumulator.inc(accumulator, Counter.HITS); - blackhole.consume(Accumulator.accumulateAndReset(accumulator)); + Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); + blackhole.consume(Accumulator.EmbeddingSupport.accumulateAndReset(accumulator)); + } + + /** + * The realistic counterpart to {@code accumulatorAccumulateAnd_highContention}: many writer + * threads incrementing, and a single dedicated thread polling {@link + * Accumulator.EmbeddingSupport#accumulateAndReset} -- not every thread doing both on every op. + * {@code accumulatorMixed-write} measures increment cost while a drain is actively contending for + * stripe locks; {@code accumulatorMixed-drain} measures the drain's own cost under that same live + * write pressure. The 4:1 writer:drainer ratio is illustrative of "many writers, rare drain," not + * tuned to a specific core count. + */ + @Benchmark + @Group("accumulatorMixed") + @GroupThreads(4) + public void accumulatorMixed_write() { + Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); + } + + @Benchmark + @Group("accumulatorMixed") + @GroupThreads(1) + public void accumulatorMixed_drain(Blackhole blackhole) { + blackhole.consume(Accumulator.EmbeddingSupport.accumulateAndReset(accumulator)); } @Benchmark diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java index 2371419b2d8..8b47237512f 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java @@ -28,7 +28,7 @@ * they are on the telemetry paths this class targets. * *

Measured on a 10-CPU machine (JDK 1.8.0_382 Zulu), 4 counters, {@code - * Accumulator.stripeCount()} = 16: + * Accumulator.EmbeddingSupport.stripeCount()} = 16: * *

{@code
  * fresh:      4 LongAdders =    160 bytes, Accumulator = 2384 bytes
@@ -78,7 +78,7 @@ static LongAdder[] freshAdders() {
   @Test
   void freshFootprint() {
     LongAdder[] adders = freshAdders();
-    long[][] accumulator = Accumulator.create(Counters.values());
+    long[][] accumulator = Accumulator.EmbeddingSupport.create(Counters.values());
 
     long adderBytes = bytes((Object) adders);
     long accumulatorBytes = bytes(accumulator);
@@ -133,7 +133,7 @@ void contendedFootprint() throws InterruptedException {
     }
 
     long contendedAdderBytes = bytes((Object) adders);
-    long[][] accumulator = Accumulator.create(Counters.values());
+    long[][] accumulator = Accumulator.EmbeddingSupport.create(Counters.values());
     long accumulatorBytes = bytes(accumulator);
 
     System.out.printf(
@@ -143,5 +143,8 @@ void contendedFootprint() throws InterruptedException {
     assertTrue(
         contendedAdderBytes >= freshAdderBytes,
         "contended LongAdder footprint should never shrink below the fresh footprint");
+    assertTrue(
+        accumulatorBytes < contendedAdderBytes,
+        "Accumulator's fixed footprint should be smaller than N contended LongAdders");
   }
 }
diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java
index 2ddff79024f..ef7c2d5706e 100644
--- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java
+++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java
@@ -4,9 +4,12 @@
 import static org.junit.jupiter.api.Assertions.assertTrue;
 
 import java.util.concurrent.CountDownLatch;
+import java.util.concurrent.ExecutionException;
 import java.util.concurrent.ExecutorService;
 import java.util.concurrent.Executors;
+import java.util.concurrent.Future;
 import java.util.concurrent.TimeUnit;
+import java.util.concurrent.TimeoutException;
 import java.util.concurrent.atomic.AtomicBoolean;
 import org.junit.jupiter.api.Test;
 
@@ -20,8 +23,8 @@ enum Counters {
 
   @Test
   void freshAccumulatorSumsToZero() {
-    long[][] data = Accumulator.create(Counters.values());
-    long[] drained = Accumulator.accumulateAndReset(data);
+    long[][] data = Accumulator.EmbeddingSupport.create(Counters.values());
+    long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data);
     for (Counters c : Counters.values()) {
       assertEquals(0L, drained[c.ordinal()]);
     }
@@ -29,12 +32,12 @@ void freshAccumulatorSumsToZero() {
 
   @Test
   void incIncrementsByOne() {
-    long[][] data = Accumulator.create(Counters.values());
-    Accumulator.inc(data, Counters.FOO);
-    Accumulator.inc(data, Counters.FOO);
-    Accumulator.inc(data, Counters.BAR);
+    long[][] data = Accumulator.EmbeddingSupport.create(Counters.values());
+    Accumulator.EmbeddingSupport.inc(data, Counters.FOO);
+    Accumulator.EmbeddingSupport.inc(data, Counters.FOO);
+    Accumulator.EmbeddingSupport.inc(data, Counters.BAR);
 
-    long[] drained = Accumulator.accumulateAndReset(data);
+    long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data);
     assertEquals(2L, drained[Counters.FOO.ordinal()]);
     assertEquals(1L, drained[Counters.BAR.ordinal()]);
     assertEquals(0L, drained[Counters.BAZ.ordinal()]);
@@ -42,39 +45,39 @@ void incIncrementsByOne() {
 
   @Test
   void addAppliesArbitraryDelta() {
-    long[][] data = Accumulator.create(Counters.values());
-    Accumulator.add(data, Counters.BAZ, 41L);
-    Accumulator.add(data, Counters.BAZ, 1L);
+    long[][] data = Accumulator.EmbeddingSupport.create(Counters.values());
+    Accumulator.EmbeddingSupport.add(data, Counters.BAZ, 41L);
+    Accumulator.EmbeddingSupport.add(data, Counters.BAZ, 1L);
 
-    long[] drained = Accumulator.accumulateAndReset(data);
+    long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data);
     assertEquals(42L, drained[Counters.BAZ.ordinal()]);
   }
 
   @Test
   void updateAppliesSeveralOpsUnderOneLock() {
-    long[][] data = Accumulator.create(Counters.values());
-    Accumulator.update(
+    long[][] data = Accumulator.EmbeddingSupport.create(Counters.values());
+    Accumulator.EmbeddingSupport.update(
         data,
         stripe -> {
-          Accumulator.inc(stripe, Counters.FOO);
-          Accumulator.inc(stripe, Counters.FOO);
-          Accumulator.add(stripe, Counters.BAR, 5L);
+          Accumulator.EmbeddingSupport.inc(stripe, Counters.FOO);
+          Accumulator.EmbeddingSupport.inc(stripe, Counters.FOO);
+          Accumulator.EmbeddingSupport.add(stripe, Counters.BAR, 5L);
         });
 
-    long[] drained = Accumulator.accumulateAndReset(data);
+    long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data);
     assertEquals(2L, drained[Counters.FOO.ordinal()]);
     assertEquals(5L, drained[Counters.BAR.ordinal()]);
   }
 
   @Test
   void accumulateAndResetsSoASecondDrainIsZero() {
-    long[][] data = Accumulator.create(Counters.values());
-    Accumulator.inc(data, Counters.FOO);
+    long[][] data = Accumulator.EmbeddingSupport.create(Counters.values());
+    Accumulator.EmbeddingSupport.inc(data, Counters.FOO);
 
-    long[] first = Accumulator.accumulateAndReset(data);
+    long[] first = Accumulator.EmbeddingSupport.accumulateAndReset(data);
     assertEquals(1L, first[Counters.FOO.ordinal()]);
 
-    long[] second = Accumulator.accumulateAndReset(data);
+    long[] second = Accumulator.EmbeddingSupport.accumulateAndReset(data);
     for (Counters c : Counters.values()) {
       assertEquals(0L, second[c.ordinal()]);
     }
@@ -82,15 +85,15 @@ void accumulateAndResetsSoASecondDrainIsZero() {
 
   @Test
   void drainedRowsAreAllTheSameLength() {
-    long[][] data = Accumulator.create(Counters.values());
-    long[] drained = Accumulator.accumulateAndReset(data);
+    long[][] data = Accumulator.EmbeddingSupport.create(Counters.values());
+    long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data);
     assertEquals(data[0].length, drained.length);
     assertTrue(drained.length >= Counters.values().length);
   }
 
   @Test
   void concurrentIncrementsAreNotLost() throws InterruptedException {
-    long[][] data = Accumulator.create(Counters.values());
+    long[][] data = Accumulator.EmbeddingSupport.create(Counters.values());
     int threadCount = 16;
     int incrementsPerThread = 10_000;
 
@@ -104,7 +107,7 @@ void concurrentIncrementsAreNotLost() throws InterruptedException {
               try {
                 start.await();
                 for (int i = 0; i < incrementsPerThread; i++) {
-                  Accumulator.inc(data, Counters.FOO);
+                  Accumulator.EmbeddingSupport.inc(data, Counters.FOO);
                 }
               } catch (InterruptedException e) {
                 Thread.currentThread().interrupt();
@@ -119,13 +122,14 @@ void concurrentIncrementsAreNotLost() throws InterruptedException {
       pool.shutdown();
     }
 
-    long[] drained = Accumulator.accumulateAndReset(data);
+    long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data);
     assertEquals((long) threadCount * incrementsPerThread, drained[Counters.FOO.ordinal()]);
   }
 
   @Test
-  void concurrentAccumulateAndDuringWritesNeverExceedsWritten() throws InterruptedException {
-    long[][] data = Accumulator.create(Counters.values());
+  void concurrentAccumulateAndDuringWritesNeverExceedsWritten()
+      throws InterruptedException, ExecutionException, TimeoutException {
+    long[][] data = Accumulator.EmbeddingSupport.create(Counters.values());
     int threadCount = 8;
     int incrementsPerThread = 5_000;
 
@@ -135,21 +139,22 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten() throws Interrupted
     long[] runningTotal = {0L};
 
     try {
-      pool.execute(
-          () -> {
-            while (!stop.get()) {
-              long[] drained = Accumulator.accumulateAndReset(data);
-              synchronized (runningTotal) {
-                runningTotal[0] += drained[Counters.FOO.ordinal()];
-              }
-            }
-          });
+      Future drainer =
+          pool.submit(
+              () -> {
+                while (!stop.get()) {
+                  long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data);
+                  synchronized (runningTotal) {
+                    runningTotal[0] += drained[Counters.FOO.ordinal()];
+                  }
+                }
+              });
 
       for (int t = 0; t < threadCount; t++) {
         pool.execute(
             () -> {
               for (int i = 0; i < incrementsPerThread; i++) {
-                Accumulator.inc(data, Counters.FOO);
+                Accumulator.EmbeddingSupport.inc(data, Counters.FOO);
               }
               done.countDown();
             });
@@ -157,7 +162,9 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten() throws Interrupted
 
       assertTrue(done.await(30, TimeUnit.SECONDS));
       stop.set(true);
-      long[] finalDrain = Accumulator.accumulateAndReset(data);
+      drainer.get(30, TimeUnit.SECONDS);
+
+      long[] finalDrain = Accumulator.EmbeddingSupport.accumulateAndReset(data);
       synchronized (runningTotal) {
         runningTotal[0] += finalDrain[Counters.FOO.ordinal()];
       }
@@ -167,4 +174,21 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten() throws Interrupted
       pool.shutdown();
     }
   }
+
+  @Test
+  void typedWrapperDelegatesToEmbeddingSupport() {
+    Accumulator counters = Accumulator.of(Counters.values());
+    counters.inc(Counters.FOO);
+    counters.inc(Counters.FOO);
+    counters.add(Counters.BAR, 5L);
+    counters.update(
+        stripe -> {
+          Accumulator.EmbeddingSupport.inc(stripe, Counters.BAZ);
+        });
+
+    long[] drained = counters.accumulateAndReset();
+    assertEquals(2L, drained[Counters.FOO.ordinal()]);
+    assertEquals(5L, drained[Counters.BAR.ordinal()]);
+    assertEquals(1L, drained[Counters.BAZ.ordinal()]);
+  }
 }

From c9f7fc2f643986cbfb901306adcc714babf1ba3f Mon Sep 17 00:00:00 2001
From: Douglas Q Hawkins 
Date: Tue, 1 Sep 2026 14:27:25 -0400
Subject: [PATCH 08/36] Wrap Accumulator's update/accumulateAndReset in typed
 Stripe/Counts views

Restores enum-ordinal type checking inside update()'s critical section
and on accumulateAndReset()'s drained result, closing the last gap left
by the EmbeddingSupport split. Stripe is constructed fresh under the
held lock and is expected to be scalar-replaced by escape analysis for
well-behaved (small, non-capturing, non-escaping) mutators; Counts
wraps the already-drained array and is a real but infrequent
per-drain allocation.

Co-Authored-By: Claude Sonnet 5 
---
 .../java/datadog/trace/util/Accumulator.java  | 85 ++++++++++++++++---
 .../datadog/trace/util/AccumulatorTest.java   | 13 ++-
 2 files changed, 76 insertions(+), 22 deletions(-)

diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java
index 34e8b6e84ee..f6c181a2dcf 100644
--- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java
+++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java
@@ -21,12 +21,12 @@
  * Accumulator counters = Accumulator.of(MyCounters.values());
  * counters.inc(MyCounters.FOO);
  * counters.update(stripe -> {
- *   Accumulator.EmbeddingSupport.inc(stripe, MyCounters.FOO);
- *   Accumulator.EmbeddingSupport.inc(stripe, MyCounters.BAR);
+ *   stripe.inc(MyCounters.FOO);
+ *   stripe.inc(MyCounters.BAR);
  * });
  *
- * long[] drained = counters.accumulateAndReset(); // combine + reset, atomically per stripe
- * long foo = drained[MyCounters.FOO.ordinal()];
+ * Accumulator.Counts drained = counters.accumulateAndReset(); // atomically per stripe
+ * long foo = drained.get(MyCounters.FOO);
  * }
* * @see EmbeddingSupport @@ -56,26 +56,83 @@ public void add(E key, long delta) { } /** - * Runs {@code mutator} against the calling thread's stripe under a single held lock -- the escape - * hatch for performing several related updates atomically with respect to a concurrent {@link - * #accumulateAndReset}. + * Runs {@code mutator} against a typed view of the calling thread's stripe under a single held + * lock -- the escape hatch for performing several related updates atomically with respect to a + * concurrent {@link #accumulateAndReset}. * * @param mutator a strategy over the selected stripe; keep it small and non-capturing so it - * inlines into the lock's critical section + * inlines into the lock's critical section, and don't let the {@link Stripe} escape it (store + * it, return it, hand it to another thread) -- see {@link Stripe} */ @StrategyConsumer - public void update(@Strategy Consumer mutator) { - EmbeddingSupport.update(data, mutator); + public void update(@Strategy Consumer> mutator) { + long[] stripe = EmbeddingSupport.stripeOf(data); + synchronized (stripe) { + mutator.accept(new Stripe<>(stripe)); + } } /** - * Combines and resets every stripe, returning the sum. + * A typed view over one stripe, handed to an {@link #update} strategy: the same enum-ordinal type + * checking {@link Accumulator} provides at the top level, applied inside the critical section + * too. * - * @return a new array indexed by the enum's {@code ordinal()} + *

Constructed fresh under the held lock on every {@link #update} call. A well-behaved {@link + * Strategy} mutator -- small, non-capturing, and never storing or returning this object -- lets + * escape analysis prove it doesn't escape the inlined call and scalar-replace it, so no + * allocation survives to run time. Break those rules (capture it in a field, return it, hand it + * to another thread) and it degrades to a real, per-call allocation instead of a compile-time + * fiction with no correctness difference either way -- just a cost one. + */ + public static final class Stripe> { + private final long[] stripe; + + private Stripe(long[] stripe) { + this.stripe = stripe; + } + + /** Increments the counter named by {@code key} in this stripe by one. */ + public void inc(E key) { + EmbeddingSupport.inc(stripe, key); + } + + /** Adds {@code delta} to the counter named by {@code key} in this stripe. */ + public void add(E key, long delta) { + EmbeddingSupport.add(stripe, key, delta); + } + } + + /** + * Combines and resets every stripe, returning the sum as a typed view. + * + * @return the sum, keyed by the enum's {@code ordinal()} * @see EmbeddingSupport#accumulateAndReset */ - public long[] accumulateAndReset() { - return EmbeddingSupport.accumulateAndReset(data); + public Counts accumulateAndReset() { + return new Counts<>(EmbeddingSupport.accumulateAndReset(data)); + } + + /** + * A typed view over a drained {@code long[]}, returned by {@link #accumulateAndReset}: the same + * enum-ordinal type checking {@link Accumulator} provides on writes, applied to the read side + * too. + * + *

Unlike {@link Stripe}, this is expected to escape -- the caller holds and reads it after the + * call returns -- so it's a real, per-drain allocation, not a scalar-replacement candidate. + * That's fine: {@link #accumulateAndReset} runs on a reporting cadence, not per {@link + * #inc}/{@link #add} call. + */ + public static final class Counts> { + private final long[] counts; + + private Counts(long[] counts) { + this.counts = counts; + } + + /** The counter named by {@code key}. */ + public long get(E key) { + return counts[key.ordinal()]; + } } /** diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java index ef7c2d5706e..a93665cb238 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -181,14 +181,11 @@ void typedWrapperDelegatesToEmbeddingSupport() { counters.inc(Counters.FOO); counters.inc(Counters.FOO); counters.add(Counters.BAR, 5L); - counters.update( - stripe -> { - Accumulator.EmbeddingSupport.inc(stripe, Counters.BAZ); - }); + counters.update(stripe -> stripe.inc(Counters.BAZ)); - long[] drained = counters.accumulateAndReset(); - assertEquals(2L, drained[Counters.FOO.ordinal()]); - assertEquals(5L, drained[Counters.BAR.ordinal()]); - assertEquals(1L, drained[Counters.BAZ.ordinal()]); + Accumulator.Counts drained = counters.accumulateAndReset(); + assertEquals(2L, drained.get(Counters.FOO)); + assertEquals(5L, drained.get(Counters.BAR)); + assertEquals(1L, drained.get(Counters.BAZ)); } } From a8e74481cd6f300f9df9316b90d0337e4566f51f Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Tue, 1 Sep 2026 14:36:33 -0400 Subject: [PATCH 09/36] Add benchmarks for Accumulator's typed API alongside raw EmbeddingSupport Pairs typedIncrement/typedUpdate/typedAccumulateAndReset against their EmbeddingSupport equivalents so the wrapper's cost is directly visible: inc/update should track the raw calls closely (Stripe is designed to scalar-replace), while accumulateAndReset is expected to run measurably slower by roughly one small allocation per drain (Counts escapes by design). Co-Authored-By: Claude Sonnet 5 --- .../trace/util/AccumulatorBenchmark.java | 60 +++++++++++++++++++ 1 file changed, 60 insertions(+) diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java index 5b6299a7eae..90a9c04e58a 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -102,6 +102,7 @@ enum Counter { private final LongAdder adder = new LongAdder(); private final long[][] accumulator = Accumulator.EmbeddingSupport.create(Counter.values()); + private final Accumulator typedAccumulator = Accumulator.of(Counter.values()); private final ConcurrentHashMap chm = new ConcurrentHashMap<>(); private final LongAdder[] longAdderGroup = {new LongAdder()}; @@ -157,6 +158,24 @@ public void accumulatorIncrement_highContention() { Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); } + /** + * The typed {@link Accumulator} wrapper's {@link Accumulator#inc}, paired against {@code + * accumulatorIncrement*} above: same underlying {@link Accumulator.EmbeddingSupport#inc} call, + * one extra field-load indirection through the instance. Should track the raw numbers closely -- + * a divergence here would mean the indirection isn't being inlined away. + */ + @Benchmark + @Threads(1) + public void typedIncrement_lowContention() { + typedAccumulator.inc(Counter.HITS); + } + + @Benchmark + @Threads(Threads.MAX) + public void typedIncrement_highContention() { + typedAccumulator.inc(Counter.HITS); + } + @Benchmark @Threads(1) public void chmAtomicLongIncrement_lowContention() { @@ -228,6 +247,47 @@ public void accumulatorMixed_drain(Blackhole blackhole) { blackhole.consume(Accumulator.EmbeddingSupport.accumulateAndReset(accumulator)); } + /** + * {@link Accumulator#update}, paired against the raw {@link Accumulator.EmbeddingSupport#update} + * lock/dispatch it wraps: the mutator here constructs a {@link Accumulator.Stripe} under the held + * lock and immediately lets it go, which is exactly the "small, non-capturing, non-escaping" + * shape documented as a scalar-replacement candidate. If escape analysis is doing its job, this + * tracks the raw call closely; if it regresses (e.g. after a JIT/JDK change, or a mutator shape + * that stops inlining), this is the number that would move. + */ + @Benchmark + @Threads(1) + public void typedUpdate_lowContention() { + typedAccumulator.update(stripe -> stripe.inc(Counter.HITS)); + } + + @Benchmark + @Threads(Threads.MAX) + public void typedUpdate_highContention() { + typedAccumulator.update(stripe -> stripe.inc(Counter.HITS)); + } + + /** + * {@link Accumulator#accumulateAndReset}, paired against the raw {@link + * Accumulator.EmbeddingSupport#accumulateAndReset} it wraps. Unlike {@link Accumulator.Stripe}, + * {@link Accumulator.Counts} is documented to escape (the caller holds and reads it after + * return), so this is expected to run measurably slower than the raw call by roughly one small + * object allocation per drain -- not a scalar-replacement candidate, and not meant to look free. + */ + @Benchmark + @Threads(1) + public void typedAccumulateAndReset_lowContention(Blackhole blackhole) { + typedAccumulator.inc(Counter.HITS); + blackhole.consume(typedAccumulator.accumulateAndReset()); + } + + @Benchmark + @Threads(Threads.MAX) + public void typedAccumulateAndReset_highContention(Blackhole blackhole) { + typedAccumulator.inc(Counter.HITS); + blackhole.consume(typedAccumulator.accumulateAndReset()); + } + @Benchmark @Threads(1) public void longAdderGroupIncrement_lowContention() { From bf67e0b7cbfcdc295b631e4fc5e83375ed9dd0a5 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Tue, 1 Sep 2026 19:17:46 -0400 Subject: [PATCH 10/36] Record typed-vs-raw Accumulator benchmark results in AccumulatorBenchmark javadoc Confirms the wrapper's cost is not measurable: typedIncrement/typedUpdate track EmbeddingSupport within noise (the Stripe scalar-replaces as designed), and typedAccumulateAndReset tracks the raw drain within noise at low contention (the Counts allocation doesn't show up at this granularity). Co-Authored-By: Claude Sonnet 5 --- .../trace/util/AccumulatorBenchmark.java | 27 +++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java index 90a9c04e58a..aec7b17333d 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -87,6 +87,33 @@ * write traffic concentrates on a few hot counters (favors thread-sharding) or spreads across many * (favors counter-sharding) -- not measured here, and worth checking against the real migration * targets before treating either number as the general answer. + * + *

{@code typed*}: what does the {@link Accumulator}/{@link Accumulator.Stripe}/{@link + * Accumulator.Counts} wrapping actually cost over calling {@link Accumulator.EmbeddingSupport} + * directly? {@code typedIncrement}/{@code typedUpdate} pair against {@code + * accumulatorIncrement} (the same underlying call), and {@code typedAccumulateAndReset} pairs + * against {@code accumulatorAccumulateAnd}. + * AccumulatorBenchmark.accumulatorIncrement_lowContention avgt 6 0.010 ± 0.001 us/op + * AccumulatorBenchmark.typedIncrement_lowContention avgt 6 0.010 ± 0.001 us/op + * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 6 0.033 ± 0.008 us/op + * AccumulatorBenchmark.typedIncrement_highContention avgt 6 0.025 ± 0.015 us/op + * AccumulatorBenchmark.typedUpdate_lowContention avgt 6 0.010 ± 0.001 us/op + * AccumulatorBenchmark.typedUpdate_highContention avgt 6 0.037 ± 0.017 us/op + * AccumulatorBenchmark.accumulatorAccumulateAnd_lowContention avgt 6 0.161 ± 0.003 us/op + * AccumulatorBenchmark.typedAccumulateAndReset_lowContention avgt 6 0.164 ± 0.005 us/op + * AccumulatorBenchmark.accumulatorAccumulateAnd_highContention avgt 6 17.399 ± 3.091 us/op + * AccumulatorBenchmark.typedAccumulateAndReset_highContention avgt 6 12.666 ± 1.272 us/op + * {@code typedIncrement}/{@code typedUpdate} track the raw calls within noise at both + * contention levels -- the field-load indirection through the {@link Accumulator} instance and the + * fresh {@link Accumulator.Stripe} constructed under {@code update}'s held lock both disappear, + * consistent with a small, non-capturing mutator letting escape analysis scalar-replace the {@code + * Stripe}. {@code typedAccumulateAndReset} also tracks the raw drain within noise at low contention + * (0.164 vs 0.161 us/op) -- the one-{@link Accumulator.Counts}-object-per-drain allocation it's + * documented to pay doesn't show up at this granularity. The high-contention gap in the other + * direction (12.666 vs 17.399) is not a real typed-vs-raw effect -- wrapping an already-drained + * array can only add cost, never remove it -- it's the same run-to-run lock-contention noise this + * exact measurement already shows above (13.431, 15.515, 17.399 us/op across three otherwise + * identical runs). Net: the wrapper's cost was not measurable in this run. */ @State(Scope.Benchmark) @Warmup(iterations = 1, time = 10) From 3484ed7ff67e6d5a4302e55b87b1943b09346422 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Tue, 1 Sep 2026 21:29:56 -0400 Subject: [PATCH 11/36] Rename accumulatorAccumulateAnd* benchmarks, cap footprint test thread count The benchmark calls EmbeddingSupport.accumulateAndReset, not accumulateAnd -- rename to match. Also cap contendedFootprint's thread count at 16: uncapped availableProcessors() on a high-core build agent spins up one thread per core, all busy-spinning for two seconds, which can dominate the host during a parallel test run for no added signal over a fixed small contention level. Co-Authored-By: Claude Sonnet 5 --- .../trace/util/AccumulatorBenchmark.java | 18 +++++++++--------- .../trace/util/AccumulatorFootprintTest.java | 2 +- 2 files changed, 10 insertions(+), 10 deletions(-) diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java index aec7b17333d..abfd778c17a 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -56,8 +56,8 @@ * AccumulatorBenchmark.chmAtomicLongIncrement_highContention avgt 6 0.417 ± 0.543 us/op * AccumulatorBenchmark.longAdderSumThenReset_lowContention avgt 6 0.012 ± 0.001 us/op * AccumulatorBenchmark.longAdderSumThenReset_highContention avgt 6 2.433 ± 0.203 us/op - * AccumulatorBenchmark.accumulatorAccumulateAnd_lowContention avgt 6 0.162 ± 0.009 us/op - * AccumulatorBenchmark.accumulatorAccumulateAnd_highContention avgt 6 15.515 ± 4.094 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset_lowContention avgt 6 0.162 ± 0.009 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 6 15.515 ± 4.094 us/op * * *

(This run had some background noise from another session on the measurement machine; the @@ -72,7 +72,7 @@ * lock a caller takes). This closes the same reset hazard as {@link Accumulator}, but stripes by * counter instead of by thread. * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 6 0.029 ± 0.051 us/op - * AccumulatorBenchmark.accumulatorAccumulateAnd_highContention avgt 6 13.431 ± 5.876 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 6 13.431 ± 5.876 us/op * AccumulatorBenchmark.longAdderGroupIncrement_highContention avgt 6 0.294 ± 0.088 us/op * AccumulatorBenchmark.longAdderGroupAccumulateAnd_highContention avgt 6 0.549 ± 0.291 us/op * Not "similar cost" -- a clean trade-off inversion. With this benchmark's single counter, @@ -92,16 +92,16 @@ * Accumulator.Counts} wrapping actually cost over calling {@link Accumulator.EmbeddingSupport} * directly? {@code typedIncrement}/{@code typedUpdate} pair against {@code * accumulatorIncrement} (the same underlying call), and {@code typedAccumulateAndReset} pairs - * against {@code accumulatorAccumulateAnd}. + * against {@code accumulatorAccumulateAndReset}. * AccumulatorBenchmark.accumulatorIncrement_lowContention avgt 6 0.010 ± 0.001 us/op * AccumulatorBenchmark.typedIncrement_lowContention avgt 6 0.010 ± 0.001 us/op * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 6 0.033 ± 0.008 us/op * AccumulatorBenchmark.typedIncrement_highContention avgt 6 0.025 ± 0.015 us/op * AccumulatorBenchmark.typedUpdate_lowContention avgt 6 0.010 ± 0.001 us/op * AccumulatorBenchmark.typedUpdate_highContention avgt 6 0.037 ± 0.017 us/op - * AccumulatorBenchmark.accumulatorAccumulateAnd_lowContention avgt 6 0.161 ± 0.003 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset_lowContention avgt 6 0.161 ± 0.003 us/op * AccumulatorBenchmark.typedAccumulateAndReset_lowContention avgt 6 0.164 ± 0.005 us/op - * AccumulatorBenchmark.accumulatorAccumulateAnd_highContention avgt 6 17.399 ± 3.091 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 6 17.399 ± 3.091 us/op * AccumulatorBenchmark.typedAccumulateAndReset_highContention avgt 6 12.666 ± 1.272 us/op * {@code typedIncrement}/{@code typedUpdate} track the raw calls within noise at both * contention levels -- the field-load indirection through the {@link Accumulator} instance and the @@ -231,7 +231,7 @@ public void longAdderSumThenReset_highContention(Blackhole blackhole) { @Benchmark @Threads(1) - public void accumulatorAccumulateAnd_lowContention(Blackhole blackhole) { + public void accumulatorAccumulateAndReset_lowContention(Blackhole blackhole) { Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); blackhole.consume(Accumulator.EmbeddingSupport.accumulateAndReset(accumulator)); } @@ -246,13 +246,13 @@ public void accumulatorAccumulateAnd_lowContention(Blackhole blackhole) { */ @Benchmark @Threads(Threads.MAX) - public void accumulatorAccumulateAnd_highContention(Blackhole blackhole) { + public void accumulatorAccumulateAndReset_highContention(Blackhole blackhole) { Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); blackhole.consume(Accumulator.EmbeddingSupport.accumulateAndReset(accumulator)); } /** - * The realistic counterpart to {@code accumulatorAccumulateAnd_highContention}: many writer + * The realistic counterpart to {@code accumulatorAccumulateAndReset_highContention}: many writer * threads incrementing, and a single dedicated thread polling {@link * Accumulator.EmbeddingSupport#accumulateAndReset} -- not every thread doing both on every op. * {@code accumulatorMixed-write} measures increment cost while a drain is actively contending for diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java index 8b47237512f..868f2ab6d0d 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java @@ -103,7 +103,7 @@ void contendedFootprint() throws InterruptedException { LongAdder[] adders = freshAdders(); long freshAdderBytes = bytes((Object) adders); - int threads = Math.max(4, Runtime.getRuntime().availableProcessors()); + int threads = Math.min(16, Math.max(4, Runtime.getRuntime().availableProcessors())); ExecutorService pool = Executors.newFixedThreadPool(threads); CountDownLatch start = new CountDownLatch(1); CountDownLatch done = new CountDownLatch(threads); From 28de3032f32d17f290c946fef4721990c85e3250 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Wed, 2 Sep 2026 08:46:41 -0400 Subject: [PATCH 12/36] Add a non-destructive Accumulator.sum() for live diagnostic reads accumulateAndReset() resets on every drain, so it can't back a live snapshot (e.g. summary()) without racing whatever else drains on a reporting cadence -- whichever call empties the stripes first starves the other's delta. sum() combines every stripe without resetting it, giving a live, non-destructive read that can't interfere with a concurrent accumulateAndReset(). --- .../java/datadog/trace/util/Accumulator.java | 37 ++++++++++++++-- .../datadog/trace/util/AccumulatorTest.java | 44 +++++++++++++++++++ 2 files changed, 78 insertions(+), 3 deletions(-) diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index f6c181a2dcf..15398802779 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -113,9 +113,21 @@ public Counts accumulateAndReset() { } /** - * A typed view over a drained {@code long[]}, returned by {@link #accumulateAndReset}: the same - * enum-ordinal type checking {@link Accumulator} provides on writes, applied to the read side - * too. + * Combines every stripe without resetting it, returning the sum as a typed view -- a live, + * non-destructive snapshot for a diagnostic read (e.g. {@code summary()}) that must not perturb + * the delta a concurrent {@link #accumulateAndReset} on a reporting cadence is about to report. + * + * @return the sum, keyed by the enum's {@code ordinal()} + * @see EmbeddingSupport#sum(long[][]) + */ + public Counts sum() { + return new Counts<>(EmbeddingSupport.sum(data)); + } + + /** + * A typed view over a drained {@code long[]}, returned by {@link #accumulateAndReset} or {@link + * #sum}: the same enum-ordinal type checking {@link Accumulator} provides on writes, applied to + * the read side too. * *

Unlike {@link Stripe}, this is expected to escape -- the caller holds and reads it after the * call returns -- so it's a real, per-drain allocation, not a scalar-replacement candidate. @@ -282,6 +294,25 @@ public static long[] accumulateAndReset(long[][] data) { return acc; } + /** + * Combines every stripe without resetting it, returning the sum -- a live, non-destructive + * snapshot for a diagnostic read that must not perturb the delta a concurrent {@link + * #accumulateAndReset} on a reporting cadence is about to report. + * + * @return a new array the same length as one stripe's row, indexed by the enum's {@code + * ordinal()} for the positions actually in use (trailing padding positions are always zero) + * @see #accumulateAndReset(long[][]) + */ + public static long[] sum(long[][] data) { + long[] acc = new long[data[0].length]; + for (long[] stripe : data) { + synchronized (stripe) { + combine(acc, stripe); + } + } + return acc; + } + /** * {@code acc[i] += stripe[i]} for every index -- a fixed-trip-count loop C2 can auto-vectorize. */ diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java index a93665cb238..12ffbba2bee 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -175,6 +175,50 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten() } } + @Test + void sumDoesNotResetStripes() { + long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); + Accumulator.EmbeddingSupport.inc(data, Counters.FOO); + + long[] first = Accumulator.EmbeddingSupport.sum(data); + assertEquals(1L, first[Counters.FOO.ordinal()]); + + // sum() didn't reset anything, so a second sum() sees the same total + long[] second = Accumulator.EmbeddingSupport.sum(data); + assertEquals(1L, second[Counters.FOO.ordinal()]); + + // and a real drain afterwards still sees the value sum() didn't consume + long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data); + assertEquals(1L, drained[Counters.FOO.ordinal()]); + } + + @Test + void sumReflectsIncrementsMadeAfterAnEarlierSum() { + long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); + Accumulator.EmbeddingSupport.inc(data, Counters.FOO); + Accumulator.EmbeddingSupport.sum(data); + + Accumulator.EmbeddingSupport.inc(data, Counters.FOO); + long[] second = Accumulator.EmbeddingSupport.sum(data); + assertEquals(2L, second[Counters.FOO.ordinal()]); + } + + @Test + void typedWrapperSumDoesNotReset() { + Accumulator counters = Accumulator.of(Counters.values()); + counters.inc(Counters.FOO); + counters.add(Counters.BAR, 5L); + + Accumulator.Counts sum = counters.sum(); + assertEquals(1L, sum.get(Counters.FOO)); + assertEquals(5L, sum.get(Counters.BAR)); + + // still there for the real drain + Accumulator.Counts drained = counters.accumulateAndReset(); + assertEquals(1L, drained.get(Counters.FOO)); + assertEquals(5L, drained.get(Counters.BAR)); + } + @Test void typedWrapperDelegatesToEmbeddingSupport() { Accumulator counters = Accumulator.of(Counters.values()); From 3c5b7ef1a39175723c45933d6a33d7e0d56a5ebb Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Wed, 2 Sep 2026 08:49:52 -0400 Subject: [PATCH 13/36] Add Accumulator.Counts.plus() to combine a stored total with a live sum Lets a caller keep a running Counts updated by periodic drains and still answer "what's the total right now" by combining it with a fresh, non-destructive sum() -- without needing to reset or drain anything just to read a live number. --- .../java/datadog/trace/util/Accumulator.java | 13 ++++++++++++ .../datadog/trace/util/AccumulatorTest.java | 21 +++++++++++++++++++ 2 files changed, 34 insertions(+) diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index 15398802779..911c73102f0 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -145,6 +145,19 @@ private Counts(long[] counts) { public long get(E key) { return counts[key.ordinal()]; } + + /** + * Adds {@code other} to this, key by key, returning a new {@link Counts} rather than mutating + * either input -- combines a stored running total with a fresh, non-destructive {@link #sum} to + * answer "what's the live total right now" without ever resetting anything. + */ + public Counts plus(Counts other) { + long[] combined = counts.clone(); + for (int i = 0; i < combined.length; i++) { + combined[i] += other.counts[i]; + } + return new Counts<>(combined); + } } /** diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java index 12ffbba2bee..f7ae69f43e8 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -219,6 +219,27 @@ void typedWrapperSumDoesNotReset() { assertEquals(5L, drained.get(Counters.BAR)); } + @Test + void plusCombinesAStoredRunningTotalWithALiveSumWithoutMutatingEither() { + Accumulator counters = Accumulator.of(Counters.values()); + counters.inc(Counters.FOO); + counters.add(Counters.BAR, 5L); + + // drain once, e.g. as if a reporting cycle already ran and stored this total + Accumulator.Counts storedTotal = counters.accumulateAndReset(); + + // more activity happens after that drain, before the next one + counters.inc(Counters.FOO); + + Accumulator.Counts live = storedTotal.plus(counters.sum()); + assertEquals(2L, live.get(Counters.FOO)); + assertEquals(5L, live.get(Counters.BAR)); + + // neither input was mutated by combining them + assertEquals(1L, storedTotal.get(Counters.FOO)); + assertEquals(1L, counters.sum().get(Counters.FOO)); + } + @Test void typedWrapperDelegatesToEmbeddingSupport() { Accumulator counters = Accumulator.of(Counters.values()); From 93e2c2d4a95e4bf92868a683a0794ee21eadb025 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Wed, 2 Sep 2026 10:20:48 -0400 Subject: [PATCH 14/36] Add Accumulator.Counts.zero() to seed a running total without a scratch Accumulator Replaces the awkward "construct an Accumulator just to call sum() on it for zeros" pattern a consumer would otherwise need to seed a stored running total before any real drain has happened. Co-Authored-By: Claude Sonnet 5 --- .../main/java/datadog/trace/util/Accumulator.java | 11 +++++++++++ .../java/datadog/trace/util/AccumulatorTest.java | 13 +++++++++++++ 2 files changed, 24 insertions(+) diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index 911c73102f0..fbae5a5ecb5 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -141,6 +141,17 @@ private Counts(long[] counts) { this.counts = counts; } + /** + * An all-zero {@link Counts}, sized for {@code values} -- for seeding a running total before + * any real drain has happened, without needing a scratch {@link Accumulator} just to call + * {@link Accumulator#sum()} on it. + * + * @param values the enum constants naming each counter, e.g. {@code MyCounters.values()} + */ + public static > Counts zero(E[] values) { + return new Counts<>(new long[values.length]); + } + /** The counter named by {@code key}. */ public long get(E key) { return counts[key.ordinal()]; diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java index f7ae69f43e8..0c343ddb03f 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -219,6 +219,19 @@ void typedWrapperSumDoesNotReset() { assertEquals(5L, drained.get(Counters.BAR)); } + @Test + void zeroSeedsAnAllZeroCountsWithoutAScratchAccumulator() { + Accumulator.Counts zero = Accumulator.Counts.zero(Counters.values()); + assertEquals(0L, zero.get(Counters.FOO)); + assertEquals(0L, zero.get(Counters.BAR)); + + Accumulator counters = Accumulator.of(Counters.values()); + counters.inc(Counters.FOO); + + Accumulator.Counts live = zero.plus(counters.sum()); + assertEquals(1L, live.get(Counters.FOO)); + } + @Test void plusCombinesAStoredRunningTotalWithALiveSumWithoutMutatingEither() { Accumulator counters = Accumulator.of(Counters.values()); From 34f1830e0ad0bfd76bcd4998805d124ac52346c9 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Wed, 2 Sep 2026 10:31:33 -0400 Subject: [PATCH 15/36] Add a contextual Accumulator.update(context, BiConsumer) overload Lets a caller pass a value the mutator needs as an explicit parameter instead of capturing it, for cases where that capture is the only thing stopping the lambda from being non-capturing (and thus allocation-free per Java's own lambda-caching behavior). Boxes the context if it's a primitive at the call site -- a real trade against the capture it replaces, not a free win. Co-Authored-By: Claude Sonnet 5 --- .../java/datadog/trace/util/Accumulator.java | 22 +++++++++++++++++++ .../datadog/trace/util/AccumulatorTest.java | 16 ++++++++++++++ 2 files changed, 38 insertions(+) diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index fbae5a5ecb5..f0ba8755974 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -4,6 +4,7 @@ import datadog.trace.api.function.Strategy; import datadog.trace.api.function.StrategyConsumer; import java.util.Arrays; +import java.util.function.BiConsumer; import java.util.function.Consumer; import javax.annotation.ParametersAreNonnullByDefault; import javax.annotation.concurrent.GuardedBy; @@ -72,6 +73,27 @@ public void update(@Strategy Consumer> mutator) { } } + /** + * Like {@link #update(Consumer)}, but passes {@code context} to {@code mutator} as an explicit + * parameter instead of letting the mutator capture it -- for a caller that would otherwise need + * to close over a local (e.g. a count) just to get it into the critical section. Note {@code + * context} is boxed if it's a primitive at the call site; that's a real allocation trade against + * the capturing lambda it replaces, not a free win -- prefer this only when {@code context} would + * otherwise be the only thing forcing a capture. + * + * @param context a value the mutator needs, passed in rather than captured + * @param mutator a strategy over {@code context} and the selected stripe; keep it small and + * non-capturing so it inlines into the lock's critical section, and don't let the {@link + * Stripe} escape it (store it, return it, hand it to another thread) -- see {@link Stripe} + */ + @StrategyConsumer + public void update(C context, @Strategy BiConsumer> mutator) { + long[] stripe = EmbeddingSupport.stripeOf(data); + synchronized (stripe) { + mutator.accept(context, new Stripe<>(stripe)); + } + } + /** * A typed view over one stripe, handed to an {@link #update} strategy: the same enum-ordinal type * checking {@link Accumulator} provides at the top level, applied inside the critical section diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java index 0c343ddb03f..f2b2ff20471 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -266,4 +266,20 @@ void typedWrapperDelegatesToEmbeddingSupport() { assertEquals(5L, drained.get(Counters.BAR)); assertEquals(1L, drained.get(Counters.BAZ)); } + + @Test + void contextualUpdatePassesContextInsteadOfCapturingIt() { + Accumulator counters = Accumulator.of(Counters.values()); + + counters.update( + 5L, + (delta, stripe) -> { + stripe.inc(Counters.FOO); + stripe.add(Counters.BAR, delta); + }); + + Accumulator.Counts drained = counters.accumulateAndReset(); + assertEquals(1L, drained.get(Counters.FOO)); + assertEquals(5L, drained.get(Counters.BAR)); + } } From 8df6143d9c53dc6b6f16868097cc0f1e25313d23 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Wed, 2 Sep 2026 11:15:19 -0400 Subject: [PATCH 16/36] Let Counts expose its own keys and add Class-based factories Accumulator/Counts now remember the enum's values() array from construction, so Counts.values() lets a caller iterate its own keys without separately threading E.values() alongside it (motivated by reducing StatsDCountReporter's call-site ceremony). Also add Accumulator.of(Class)/Counts.zero(Class) overloads alongside the existing array-based ones, for callers that'd rather pass MyEnum.class. Co-Authored-By: Claude Sonnet 5 --- .../java/datadog/trace/util/Accumulator.java | 42 +++++++++++++++---- .../datadog/trace/util/AccumulatorTest.java | 26 ++++++++++++ 2 files changed, 61 insertions(+), 7 deletions(-) diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index f0ba8755974..11c63b44a65 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -34,16 +34,25 @@ */ public final class Accumulator> { private final long[][] data; + private final E[] values; - private Accumulator(long[][] data) { + private Accumulator(long[][] data, E[] values) { this.data = data; + this.values = values; } /** * @param values the enum constants naming each counter, e.g. {@code MyCounters.values()} */ public static > Accumulator of(E[] values) { - return new Accumulator<>(EmbeddingSupport.create(values)); + return new Accumulator<>(EmbeddingSupport.create(values), values); + } + + /** + * @param enumType the enum naming each counter, e.g. {@code MyCounters.class} + */ + public static > Accumulator of(Class enumType) { + return of(enumType.getEnumConstants()); } /** Increments the counter named by {@code key} in the calling thread's stripe by one. */ @@ -131,7 +140,7 @@ public void add(E key, long delta) { * @see EmbeddingSupport#accumulateAndReset */ public Counts accumulateAndReset() { - return new Counts<>(EmbeddingSupport.accumulateAndReset(data)); + return new Counts<>(EmbeddingSupport.accumulateAndReset(data), values); } /** @@ -143,7 +152,7 @@ public Counts accumulateAndReset() { * @see EmbeddingSupport#sum(long[][]) */ public Counts sum() { - return new Counts<>(EmbeddingSupport.sum(data)); + return new Counts<>(EmbeddingSupport.sum(data), values); } /** @@ -158,9 +167,11 @@ public Counts sum() { */ public static final class Counts> { private final long[] counts; + private final E[] values; - private Counts(long[] counts) { + private Counts(long[] counts, E[] values) { this.counts = counts; + this.values = values; } /** @@ -171,7 +182,15 @@ private Counts(long[] counts) { * @param values the enum constants naming each counter, e.g. {@code MyCounters.values()} */ public static > Counts zero(E[] values) { - return new Counts<>(new long[values.length]); + return new Counts<>(new long[values.length], values); + } + + /** + * @param enumType the enum naming each counter, e.g. {@code MyCounters.class} + * @see #zero(Enum[]) + */ + public static > Counts zero(Class enumType) { + return zero(enumType.getEnumConstants()); } /** The counter named by {@code key}. */ @@ -179,6 +198,15 @@ public long get(E key) { return counts[key.ordinal()]; } + /** + * The enum constants this {@link Counts} is keyed by, in declaration order -- for a caller that + * wants to iterate every counter (e.g. reporting each one) without separately having to pass + * {@code E.values()} alongside this object. + */ + public E[] values() { + return values; + } + /** * Adds {@code other} to this, key by key, returning a new {@link Counts} rather than mutating * either input -- combines a stored running total with a fresh, non-destructive {@link #sum} to @@ -189,7 +217,7 @@ public Counts plus(Counts other) { for (int i = 0; i < combined.length; i++) { combined[i] += other.counts[i]; } - return new Counts<>(combined); + return new Counts<>(combined, values); } } diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java index f2b2ff20471..c99c4683076 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -232,6 +232,32 @@ void zeroSeedsAnAllZeroCountsWithoutAScratchAccumulator() { assertEquals(1L, live.get(Counters.FOO)); } + @Test + void ofAndZeroAcceptAnEnumClassInsteadOfAValuesArray() { + Accumulator counters = Accumulator.of(Counters.class); + counters.inc(Counters.FOO); + + Accumulator.Counts zero = Accumulator.Counts.zero(Counters.class); + Accumulator.Counts live = zero.plus(counters.sum()); + assertEquals(1L, live.get(Counters.FOO)); + } + + @Test + void countsExposesItsOwnKeysWithoutASeparateValuesArray() { + Accumulator counters = Accumulator.of(Counters.values()); + counters.inc(Counters.FOO); + counters.add(Counters.BAR, 5L); + + Accumulator.Counts drained = counters.accumulateAndReset(); + assertEquals(Counters.values().length, drained.values().length); + + long total = 0L; + for (Counters c : drained.values()) { + total += drained.get(c); + } + assertEquals(6L, total); + } + @Test void plusCombinesAStoredRunningTotalWithALiveSumWithoutMutatingEither() { Accumulator counters = Accumulator.of(Counters.values()); From 9bc7abccf616a9fde3310a5d1236c78b6abe769a Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Wed, 2 Sep 2026 11:21:30 -0400 Subject: [PATCH 17/36] Rename Counts.values() to Counts.keys() values() read as reusing Enum.values()'s exact word for a different concept -- Counts is a generic primitive, not metrics-specific, so keys() (matching the existing get(E key) naming) is clearer without tying the name to any one consumer. Co-Authored-By: Claude Sonnet 5 --- .../src/main/java/datadog/trace/util/Accumulator.java | 2 +- .../src/test/java/datadog/trace/util/AccumulatorTest.java | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index 11c63b44a65..b045506dfa8 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -203,7 +203,7 @@ public long get(E key) { * wants to iterate every counter (e.g. reporting each one) without separately having to pass * {@code E.values()} alongside this object. */ - public E[] values() { + public E[] keys() { return values; } diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java index c99c4683076..fa10668e4d3 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -249,10 +249,10 @@ void countsExposesItsOwnKeysWithoutASeparateValuesArray() { counters.add(Counters.BAR, 5L); Accumulator.Counts drained = counters.accumulateAndReset(); - assertEquals(Counters.values().length, drained.values().length); + assertEquals(Counters.values().length, drained.keys().length); long total = 0L; - for (Counters c : drained.values()) { + for (Counters c : drained.keys()) { total += drained.get(c); } assertEquals(6L, total); From 7e80a21574ea640f58d6921f7ebd034f68467b92 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Wed, 2 Sep 2026 11:57:46 -0400 Subject: [PATCH 18/36] Add Accumulator.update(long, ObjLongConsumer) to avoid boxing a primitive context --- .../java/datadog/trace/util/Accumulator.java | 31 +++++++++++++++- .../datadog/trace/util/AccumulatorTest.java | 36 ++++++++++++++++++- 2 files changed, 65 insertions(+), 2 deletions(-) diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index b045506dfa8..2e247b2e981 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -6,6 +6,7 @@ import java.util.Arrays; import java.util.function.BiConsumer; import java.util.function.Consumer; +import java.util.function.ObjLongConsumer; import javax.annotation.ParametersAreNonnullByDefault; import javax.annotation.concurrent.GuardedBy; @@ -88,7 +89,8 @@ public void update(@Strategy Consumer> mutator) { * to close over a local (e.g. a count) just to get it into the critical section. Note {@code * context} is boxed if it's a primitive at the call site; that's a real allocation trade against * the capturing lambda it replaces, not a free win -- prefer this only when {@code context} would - * otherwise be the only thing forcing a capture. + * otherwise be the only thing forcing a capture. For an {@code int} or {@code long} context, use + * {@link #update(long, ObjLongConsumer)} instead to avoid that boxing entirely. * * @param context a value the mutator needs, passed in rather than captured * @param mutator a strategy over {@code context} and the selected stripe; keep it small and @@ -103,6 +105,33 @@ public void update(C context, @Strategy BiConsumer> mutator) { } } + /** + * Like {@link #update(Object, BiConsumer)}, but for a {@code long} context -- reuses the JDK's + * {@link ObjLongConsumer} instead of the generic {@link BiConsumer}, so {@code context} is passed + * as a primitive {@code long} rather than boxed into a {@link Long}. Covers an {@code int} + * context too: it widens to {@code long} for free at the call site, no boxing either way. (A + * dedicated {@code int} overload isn't offered alongside this one -- an {@code int} argument + * would be ambiguous between the two, since it's an exact match for one and a free widening + * conversion to the other, and unrelated functional-interface types block the usual most-specific + * tiebreak.) + * + *

Note the parameter order this forces: {@link ObjLongConsumer#accept} takes {@code (T, + * long)}, so the mutator sees the stripe first and the context second -- the opposite order from + * {@link #update(Object, BiConsumer)}. + * + * @param context a primitive value the mutator needs, passed in rather than captured or boxed + * @param mutator a strategy over the selected stripe and {@code context}; keep it small and + * non-capturing so it inlines into the lock's critical section, and don't let the {@link + * Stripe} escape it (store it, return it, hand it to another thread) -- see {@link Stripe} + */ + @StrategyConsumer + public void update(long context, @Strategy ObjLongConsumer> mutator) { + long[] stripe = EmbeddingSupport.stripeOf(data); + synchronized (stripe) { + mutator.accept(new Stripe<>(stripe), context); + } + } + /** * A typed view over one stripe, handed to an {@link #update} strategy: the same enum-ordinal type * checking {@link Accumulator} provides at the top level, applied inside the critical section diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java index fa10668e4d3..5caa32549ed 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -296,10 +296,44 @@ void typedWrapperDelegatesToEmbeddingSupport() { @Test void contextualUpdatePassesContextInsteadOfCapturingIt() { Accumulator counters = Accumulator.of(Counters.values()); + String context = "abcde"; + + counters.update( + context, + (ctx, stripe) -> { + stripe.inc(Counters.FOO); + stripe.add(Counters.BAR, ctx.length()); + }); + + Accumulator.Counts drained = counters.accumulateAndReset(); + assertEquals(1L, drained.get(Counters.FOO)); + assertEquals(5L, drained.get(Counters.BAR)); + } + + @Test + void intContextWidensIntoTheLongOverloadWithoutBoxing() { + Accumulator counters = Accumulator.of(Counters.values()); + int delta = 5; + + counters.update( + delta, + (stripe, d) -> { + stripe.inc(Counters.FOO); + stripe.add(Counters.BAR, d); + }); + + Accumulator.Counts drained = counters.accumulateAndReset(); + assertEquals(1L, drained.get(Counters.FOO)); + assertEquals(5L, drained.get(Counters.BAR)); + } + + @Test + void longContextualUpdateAvoidsBoxing() { + Accumulator counters = Accumulator.of(Counters.values()); counters.update( 5L, - (delta, stripe) -> { + (stripe, delta) -> { stripe.inc(Counters.FOO); stripe.add(Counters.BAR, delta); }); From 69251a58c352f0dafdbf6dd8ff2bb0131f6666e0 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Wed, 2 Sep 2026 14:15:04 -0400 Subject: [PATCH 19/36] Only touch the real counter width in combine/reset, not the padded stripe row accumulateAndReset()/sum() drained the full paddedWidth-length stripe row, so every drain wrote through Arrays.fill into the trailing cache-line buffer that paddedWidth() adds to keep adjacent stripes from false-sharing. Thread the real width through combine()/reset() so the padding stays untouched after create()'s zero-init, matching the array size Counts.zero() already returns. --- .../trace/util/AccumulatorBenchmark.java | 9 ++-- .../java/datadog/trace/util/Accumulator.java | 54 +++++++++++-------- .../datadog/trace/util/AccumulatorTest.java | 46 +++++++++------- 3 files changed, 66 insertions(+), 43 deletions(-) diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java index abfd778c17a..3b7c7eeb05f 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -233,7 +233,8 @@ public void longAdderSumThenReset_highContention(Blackhole blackhole) { @Threads(1) public void accumulatorAccumulateAndReset_lowContention(Blackhole blackhole) { Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); - blackhole.consume(Accumulator.EmbeddingSupport.accumulateAndReset(accumulator)); + blackhole.consume( + Accumulator.EmbeddingSupport.accumulateAndReset(accumulator, Counter.values().length)); } /** @@ -248,7 +249,8 @@ public void accumulatorAccumulateAndReset_lowContention(Blackhole blackhole) { @Threads(Threads.MAX) public void accumulatorAccumulateAndReset_highContention(Blackhole blackhole) { Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); - blackhole.consume(Accumulator.EmbeddingSupport.accumulateAndReset(accumulator)); + blackhole.consume( + Accumulator.EmbeddingSupport.accumulateAndReset(accumulator, Counter.values().length)); } /** @@ -271,7 +273,8 @@ public void accumulatorMixed_write() { @Group("accumulatorMixed") @GroupThreads(1) public void accumulatorMixed_drain(Blackhole blackhole) { - blackhole.consume(Accumulator.EmbeddingSupport.accumulateAndReset(accumulator)); + blackhole.consume( + Accumulator.EmbeddingSupport.accumulateAndReset(accumulator, Counter.values().length)); } /** diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index 2e247b2e981..746c985a5b6 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -169,7 +169,7 @@ public void add(E key, long delta) { * @see EmbeddingSupport#accumulateAndReset */ public Counts accumulateAndReset() { - return new Counts<>(EmbeddingSupport.accumulateAndReset(data), values); + return new Counts<>(EmbeddingSupport.accumulateAndReset(data, values.length), values); } /** @@ -178,10 +178,10 @@ public Counts accumulateAndReset() { * the delta a concurrent {@link #accumulateAndReset} on a reporting cadence is about to report. * * @return the sum, keyed by the enum's {@code ordinal()} - * @see EmbeddingSupport#sum(long[][]) + * @see EmbeddingSupport#sum(long[][], int) */ public Counts sum() { - return new Counts<>(EmbeddingSupport.sum(data), values); + return new Counts<>(EmbeddingSupport.sum(data, values.length), values); } /** @@ -285,7 +285,8 @@ public Counts plus(Counts other) { * Accumulator.EmbeddingSupport.inc(stripe, MyCounters.BAR); * }); * - * long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data); // per stripe + * long[] drained = + * Accumulator.EmbeddingSupport.accumulateAndReset(data, MyCounters.values().length); * long foo = drained[MyCounters.FOO.ordinal()]; * } */ @@ -383,15 +384,20 @@ public static void update(long[][] data, @Strategy Consumer mutator) { * {@link #inc}/{@link #add}/{@link #update} -- so no writer can land an increment in the gap * between summing and zeroing the way {@code LongAdder#sumThenReset()} allows. * - * @return a new array the same length as one stripe's row, indexed by the enum's {@code - * ordinal()} for the positions actually in use (trailing padding positions are always zero) + *

Only the first {@code width} positions of each stripe are read or written -- the trailing + * cache line {@link #paddedWidth} reserves past that point is never touched again after {@link + * #create} zero-initializes it, so it stays a genuinely dead buffer between adjacent stripe + * rows instead of being read-and-rewritten (dirtying that cache line) on every drain. + * + * @param width the number of counters actually in use, e.g. {@code values.length} + * @return a new array of length {@code width}, indexed by the enum's {@code ordinal()} */ - public static long[] accumulateAndReset(long[][] data) { - long[] acc = new long[data[0].length]; + public static long[] accumulateAndReset(long[][] data, int width) { + long[] acc = new long[width]; for (long[] stripe : data) { synchronized (stripe) { - combine(acc, stripe); - reset(stripe); + combine(acc, stripe, width); + reset(stripe, width); } } return acc; @@ -402,34 +408,38 @@ public static long[] accumulateAndReset(long[][] data) { * snapshot for a diagnostic read that must not perturb the delta a concurrent {@link * #accumulateAndReset} on a reporting cadence is about to report. * - * @return a new array the same length as one stripe's row, indexed by the enum's {@code - * ordinal()} for the positions actually in use (trailing padding positions are always zero) - * @see #accumulateAndReset(long[][]) + * @param width the number of counters actually in use, e.g. {@code values.length} + * @return a new array of length {@code width}, indexed by the enum's {@code ordinal()} + * @see #accumulateAndReset(long[][], int) */ - public static long[] sum(long[][] data) { - long[] acc = new long[data[0].length]; + public static long[] sum(long[][] data, int width) { + long[] acc = new long[width]; for (long[] stripe : data) { synchronized (stripe) { - combine(acc, stripe); + combine(acc, stripe, width); } } return acc; } /** - * {@code acc[i] += stripe[i]} for every index -- a fixed-trip-count loop C2 can auto-vectorize. + * {@code acc[i] += stripe[i]} for {@code i} in {@code [0, width)} -- a fixed-trip-count loop C2 + * can auto-vectorize. */ @GuardedBy("stripe") - private static void combine(long[] acc, long[] stripe) { - for (int i = 0; i < acc.length; i++) { + private static void combine(long[] acc, long[] stripe, int width) { + for (int i = 0; i < width; i++) { acc[i] += stripe[i]; } } - /** Zeroes every position of {@code stripe}, via the JVM-intrinsic {@link Arrays#fill}. */ + /** + * Zeroes {@code stripe}'s first {@code width} positions, via the JVM-intrinsic {@link + * Arrays#fill}. Deliberately stops at {@code width}, leaving the trailing padding untouched. + */ @GuardedBy("stripe") - private static void reset(long[] stripe) { - Arrays.fill(stripe, 0L); + private static void reset(long[] stripe, int width) { + Arrays.fill(stripe, 0, width, 0L); } /** diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java index 5caa32549ed..a767b803476 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -24,7 +24,8 @@ enum Counters { @Test void freshAccumulatorSumsToZero() { long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); - long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data); + long[] drained = + Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); for (Counters c : Counters.values()) { assertEquals(0L, drained[c.ordinal()]); } @@ -37,7 +38,8 @@ void incIncrementsByOne() { Accumulator.EmbeddingSupport.inc(data, Counters.FOO); Accumulator.EmbeddingSupport.inc(data, Counters.BAR); - long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data); + long[] drained = + Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); assertEquals(2L, drained[Counters.FOO.ordinal()]); assertEquals(1L, drained[Counters.BAR.ordinal()]); assertEquals(0L, drained[Counters.BAZ.ordinal()]); @@ -49,7 +51,8 @@ void addAppliesArbitraryDelta() { Accumulator.EmbeddingSupport.add(data, Counters.BAZ, 41L); Accumulator.EmbeddingSupport.add(data, Counters.BAZ, 1L); - long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data); + long[] drained = + Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); assertEquals(42L, drained[Counters.BAZ.ordinal()]); } @@ -64,7 +67,8 @@ void updateAppliesSeveralOpsUnderOneLock() { Accumulator.EmbeddingSupport.add(stripe, Counters.BAR, 5L); }); - long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data); + long[] drained = + Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); assertEquals(2L, drained[Counters.FOO.ordinal()]); assertEquals(5L, drained[Counters.BAR.ordinal()]); } @@ -74,21 +78,22 @@ void accumulateAndResetsSoASecondDrainIsZero() { long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); Accumulator.EmbeddingSupport.inc(data, Counters.FOO); - long[] first = Accumulator.EmbeddingSupport.accumulateAndReset(data); + long[] first = Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); assertEquals(1L, first[Counters.FOO.ordinal()]); - long[] second = Accumulator.EmbeddingSupport.accumulateAndReset(data); + long[] second = Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); for (Counters c : Counters.values()) { assertEquals(0L, second[c.ordinal()]); } } @Test - void drainedRowsAreAllTheSameLength() { + void drainedArrayIsExactlyWidthLong() { long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); - long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data); - assertEquals(data[0].length, drained.length); - assertTrue(drained.length >= Counters.values().length); + long[] drained = + Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); + assertEquals(Counters.values().length, drained.length); + assertTrue(drained.length < data[0].length, "drained array should exclude stripe padding"); } @Test @@ -122,7 +127,8 @@ void concurrentIncrementsAreNotLost() throws InterruptedException { pool.shutdown(); } - long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data); + long[] drained = + Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); assertEquals((long) threadCount * incrementsPerThread, drained[Counters.FOO.ordinal()]); } @@ -143,7 +149,9 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten() pool.submit( () -> { while (!stop.get()) { - long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data); + long[] drained = + Accumulator.EmbeddingSupport.accumulateAndReset( + data, Counters.values().length); synchronized (runningTotal) { runningTotal[0] += drained[Counters.FOO.ordinal()]; } @@ -164,7 +172,8 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten() stop.set(true); drainer.get(30, TimeUnit.SECONDS); - long[] finalDrain = Accumulator.EmbeddingSupport.accumulateAndReset(data); + long[] finalDrain = + Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); synchronized (runningTotal) { runningTotal[0] += finalDrain[Counters.FOO.ordinal()]; } @@ -180,15 +189,16 @@ void sumDoesNotResetStripes() { long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); Accumulator.EmbeddingSupport.inc(data, Counters.FOO); - long[] first = Accumulator.EmbeddingSupport.sum(data); + long[] first = Accumulator.EmbeddingSupport.sum(data, Counters.values().length); assertEquals(1L, first[Counters.FOO.ordinal()]); // sum() didn't reset anything, so a second sum() sees the same total - long[] second = Accumulator.EmbeddingSupport.sum(data); + long[] second = Accumulator.EmbeddingSupport.sum(data, Counters.values().length); assertEquals(1L, second[Counters.FOO.ordinal()]); // and a real drain afterwards still sees the value sum() didn't consume - long[] drained = Accumulator.EmbeddingSupport.accumulateAndReset(data); + long[] drained = + Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); assertEquals(1L, drained[Counters.FOO.ordinal()]); } @@ -196,10 +206,10 @@ void sumDoesNotResetStripes() { void sumReflectsIncrementsMadeAfterAnEarlierSum() { long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); Accumulator.EmbeddingSupport.inc(data, Counters.FOO); - Accumulator.EmbeddingSupport.sum(data); + Accumulator.EmbeddingSupport.sum(data, Counters.values().length); Accumulator.EmbeddingSupport.inc(data, Counters.FOO); - long[] second = Accumulator.EmbeddingSupport.sum(data); + long[] second = Accumulator.EmbeddingSupport.sum(data, Counters.values().length); assertEquals(2L, second[Counters.FOO.ordinal()]); } From baf0a3377ecb54f3f1c017a2930483aae7e23520 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Wed, 2 Sep 2026 17:14:12 -0400 Subject: [PATCH 20/36] Replace Accumulator's synchronized stripes with lock-free AtomicLongArray striping Benchmarking showed the AtomicLongArray design beats the realistic LongAdder-per-counter baseline by ~2 orders of magnitude on increment (the hot path) at a small cost on drain (the rare, periodic path). Per-counter atomicity is preserved via getAndSet/getAndAdd; row-wide atomicity across counters is dropped since no real caller needs it, so EmbeddingSupport, Stripe, and the update() overloads are removed. Co-Authored-By: Claude Sonnet 5 --- .../trace/util/AccumulatorBenchmark.java | 216 ++------- .../java/datadog/trace/util/Accumulator.java | 440 ++++-------------- .../trace/util/AccumulatorFootprintTest.java | 30 +- .../datadog/trace/util/AccumulatorTest.java | 204 ++------ 4 files changed, 186 insertions(+), 704 deletions(-) diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java index 3b7c7eeb05f..a1e1679c9f0 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -20,100 +20,42 @@ import org.openjdk.jmh.infra.Blackhole; /** - * {@link Accumulator} vs {@link LongAdder} vs the {@code ConcurrentHashMap.computeIfAbsent(key, k - * -> new AtomicLong())} anti-pattern, at one thread (no contention) and at {@link Threads#MAX} - * (heavy contention). The CHM variant allocates its counter under the bucket's bin lock the first - * time its one constant key is seen -- exactly the pathology {@link Accumulator} exists to avoid -- - * but since the map is a {@code @State(Scope.Benchmark)} field shared across the whole run, that - * allocation happens exactly once; every sampled op after it hits the warmed, already-present fast - * path. So this measures steady-state {@code computeIfAbsent} lookup overhead on an - * already-populated map, not the one-time allocation-under-lock cost -- still a useful number (a - * fixed, small key set that's allocated once and hit for the life of the process, as {@code + * {@link Accumulator} vs the alternatives it actually displaces: a single {@code LongAdder} (the + * collision-free baseline it can never beat, only approach), an independent {@code LongAdder} per + * counter guarded by a per-counter lock (the "just fix it with LongAdder" natural migration target + * -- {@code longAdderGroup*}), and the {@code ConcurrentHashMap.computeIfAbsent(key, k -> new + * AtomicLong())} anti-pattern ({@code chmAtomicLongIncrement*}) that {@link Accumulator} exists to + * avoid. The CHM variant allocates its counter under the bucket's bin lock the first time its one + * constant key is seen, but since the map is a {@code @State(Scope.Benchmark)} field shared across + * the whole run, that allocation happens exactly once; every sampled op after it hits the warmed, + * already-present fast path. So this measures steady-state {@code computeIfAbsent} lookup overhead + * on an already-populated map, not the one-time allocation-under-lock cost -- still a useful number + * (a fixed, small key set that's allocated once and hit for the life of the process, as {@code * WafMetricCollector}-style CHM counters are, spends nearly all its time in this same warmed path), * just not the pathology the name of this benchmark might suggest. * - *

Contention result to note: at low contention, {@code accumulatorIncrement} is - * essentially free and on par with {@code longAdderIncrement}. At {@code Threads.MAX} (10 threads - * on the measurement machine), oversizing {@link Accumulator}'s stripe count from 8 (one per core) - * to 16 (roughly 2x cores, see {@code stripeCount()}) cut {@code - * accumulatorIncrement_highContention} from ~0.097 us/op to ~0.040 us/op -- fewer threads collide - * on a stripe, so fewer of them pay {@code synchronized}'s blocking wait instead of a cheap - * fast-path lock. It is still roughly 4-5x slower than {@code longAdderIncrement} (a collision-free - * CAS retry beats even an uncontended monitor enter/exit), and {@code accumulateAndReset} under - * concurrent writers got correspondingly more expensive (~7.5us to ~15.5us) since draining now - * walks twice as many stripes while writers are actively landing on them. Read {@code - * accumulatorIncrement_highContention} not as "Accumulator beats LongAdder under contention" (it - * doesn't, on this shape) but as the honest cost of the drain-under-lock design that buys atomic - * combine+reset; a caller trading that safety for raw increment throughput should measure their own - * contention level before choosing between them. - * Apple M1 Max, 10 CPUs - JDK 1.8.0_382 (Zulu) - macOS/arm64 - stripeCount() = 16 - * Benchmark Mode Cnt Score Error Units - * AccumulatorBenchmark.longAdderIncrement_lowContention avgt 6 0.007 ± 0.001 us/op - * AccumulatorBenchmark.longAdderIncrement_highContention avgt 6 0.009 ± 0.001 us/op - * AccumulatorBenchmark.accumulatorIncrement_lowContention avgt 6 0.010 ± 0.001 us/op - * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 6 0.040 ± 0.002 us/op - * AccumulatorBenchmark.chmAtomicLongIncrement_lowContention avgt 6 0.010 ± 0.001 us/op - * AccumulatorBenchmark.chmAtomicLongIncrement_highContention avgt 6 0.417 ± 0.543 us/op - * AccumulatorBenchmark.longAdderSumThenReset_lowContention avgt 6 0.012 ± 0.001 us/op - * AccumulatorBenchmark.longAdderSumThenReset_highContention avgt 6 2.433 ± 0.203 us/op - * AccumulatorBenchmark.accumulatorAccumulateAndReset_lowContention avgt 6 0.162 ± 0.009 us/op - * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 6 15.515 ± 4.094 us/op - * - * - *

(This run had some background noise from another session on the measurement machine; the - * {@code lowContention} rows and the {@code highContention} directional deltas are reliable, but - * treat the exact {@code highContention} magnitudes as approximate.) - * *

{@code longAdderGroup*}: is a "just fix it with LongAdder" helper actually cheaper? * {@code groupInc}/{@code groupAccumulateAnd} are the natural correct fix using {@code LongAdder} * as the payload: one {@code LongAdder} per counter, with a per-counter lock guarding both * the increment and the drain (locking only the drain does nothing -- {@code sumThenReset()}'s * internal race is against the {@code LongAdder}'s own CAS-based {@code add()}, not against any - * lock a caller takes). This closes the same reset hazard as {@link Accumulator}, but stripes by - * counter instead of by thread. - * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 6 0.029 ± 0.051 us/op - * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 6 13.431 ± 5.876 us/op - * AccumulatorBenchmark.longAdderGroupIncrement_highContention avgt 6 0.294 ± 0.088 us/op - * AccumulatorBenchmark.longAdderGroupAccumulateAnd_highContention avgt 6 0.549 ± 0.291 us/op - * Not "similar cost" -- a clean trade-off inversion. With this benchmark's single counter, - * {@code longAdderGroup}'s per-counter lock collapses to one lock for every thread (no thread-based - * distribution at all), so it loses badly on the write path: ~10x worse than {@code Accumulator}'s - * thread-sharded stripes. But its drain only has that one lock to acquire, so it wins big there: - * ~24x better than {@code Accumulator}, which always walks all 16 stripes on every drain regardless - * of counter count. That asymmetry is the whole story: {@code longAdderGroup}'s drain cost scales - * with number of counters (more counters -> more locks to drain), while {@code - * Accumulator}'s drain cost is fixed at stripe count, independent of counter count. Which design - * actually wins for a given caller depends on that caller's counter cardinality and whether its - * write traffic concentrates on a few hot counters (favors thread-sharding) or spreads across many - * (favors counter-sharding) -- not measured here, and worth checking against the real migration - * targets before treating either number as the general answer. - * - *

{@code typed*}: what does the {@link Accumulator}/{@link Accumulator.Stripe}/{@link - * Accumulator.Counts} wrapping actually cost over calling {@link Accumulator.EmbeddingSupport} - * directly? {@code typedIncrement}/{@code typedUpdate} pair against {@code - * accumulatorIncrement} (the same underlying call), and {@code typedAccumulateAndReset} pairs - * against {@code accumulatorAccumulateAndReset}. - * AccumulatorBenchmark.accumulatorIncrement_lowContention avgt 6 0.010 ± 0.001 us/op - * AccumulatorBenchmark.typedIncrement_lowContention avgt 6 0.010 ± 0.001 us/op - * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 6 0.033 ± 0.008 us/op - * AccumulatorBenchmark.typedIncrement_highContention avgt 6 0.025 ± 0.015 us/op - * AccumulatorBenchmark.typedUpdate_lowContention avgt 6 0.010 ± 0.001 us/op - * AccumulatorBenchmark.typedUpdate_highContention avgt 6 0.037 ± 0.017 us/op - * AccumulatorBenchmark.accumulatorAccumulateAndReset_lowContention avgt 6 0.161 ± 0.003 us/op - * AccumulatorBenchmark.typedAccumulateAndReset_lowContention avgt 6 0.164 ± 0.005 us/op - * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 6 17.399 ± 3.091 us/op - * AccumulatorBenchmark.typedAccumulateAndReset_highContention avgt 6 12.666 ± 1.272 us/op - * {@code typedIncrement}/{@code typedUpdate} track the raw calls within noise at both - * contention levels -- the field-load indirection through the {@link Accumulator} instance and the - * fresh {@link Accumulator.Stripe} constructed under {@code update}'s held lock both disappear, - * consistent with a small, non-capturing mutator letting escape analysis scalar-replace the {@code - * Stripe}. {@code typedAccumulateAndReset} also tracks the raw drain within noise at low contention - * (0.164 vs 0.161 us/op) -- the one-{@link Accumulator.Counts}-object-per-drain allocation it's - * documented to pay doesn't show up at this granularity. The high-contention gap in the other - * direction (12.666 vs 17.399) is not a real typed-vs-raw effect -- wrapping an already-drained - * array can only add cost, never remove it -- it's the same run-to-run lock-contention noise this - * exact measurement already shows above (13.431, 15.515, 17.399 us/op across three otherwise - * identical runs). Net: the wrapper's cost was not measurable in this run. + * lock a caller takes). With this benchmark's single counter, that per-counter lock collapses to + * one lock shared by every thread -- no thread-based distribution at all -- so it loses badly on + * the write path against {@link Accumulator}'s thread-sharded stripes, especially under contention. + * This is the realistic production baseline this class was built to replace (see {@code + * TracerHealthMetrics}'s pre-migration design, one {@code LongAdder} field per counter): + * AccumulatorBenchmark.accumulatorIncrement_lowContention avgt 6 0.007 ± 0.001 us/op + * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 6 0.009 ± 0.001 us/op + * AccumulatorBenchmark.longAdderGroupIncrement_lowContention avgt 6 0.012 ± 0.001 us/op + * AccumulatorBenchmark.longAdderGroupIncrement_highContention avgt 6 1.178 ± 0.134 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset_lowContention avgt 6 0.054 ± 0.001 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 6 2.740 ± 0.210 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd_lowContention avgt 6 0.024 ± 0.001 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd_highContention avgt 6 2.439 ± 0.337 us/op + * At high contention, {@link Accumulator} beats the realistic {@code longAdderGroup} + * baseline by roughly two orders of magnitude on increment (the call that runs on every event) + * while being slightly worse on drain (the call that runs once per reporting cycle) -- a clean win + * once weighted by call-site frequency, not just a wash. */ @State(Scope.Benchmark) @Warmup(iterations = 1, time = 10) @@ -128,8 +70,7 @@ enum Counter { } private final LongAdder adder = new LongAdder(); - private final long[][] accumulator = Accumulator.EmbeddingSupport.create(Counter.values()); - private final Accumulator typedAccumulator = Accumulator.of(Counter.values()); + private final Accumulator accumulator = Accumulator.of(Counter.values()); private final ConcurrentHashMap chm = new ConcurrentHashMap<>(); private final LongAdder[] longAdderGroup = {new LongAdder()}; @@ -138,10 +79,10 @@ enum Counter { * with a per-counter lock guarding both the increment and the drain -- external locking around * only the drain does nothing, since {@code sumThenReset()}'s internal race is against the {@code * LongAdder}'s own CAS-based {@code add()}, not against any lock a caller takes. This is the fair - * comparison point: it closes the same hazard {@link Accumulator} does, but stripes by - * counter (one lock per enum constant) instead of by thread (one lock per - * stripe, shared by all counters) -- so N threads hammering the *same* counter contend on one - * lock regardless of core count, with no thread-bucket distribution at all. + * comparison point: it closes the same reset hazard {@link Accumulator} does, but stripes by + * counter (one lock per enum constant) instead of by thread (one shared table + * across all counters) -- so N threads hammering the *same* counter contend on one lock + * regardless of core count, with no thread-bucket distribution at all. */ private static void groupInc(LongAdder[] group, int ordinal) { LongAdder counter = group[ordinal]; @@ -176,31 +117,13 @@ public void longAdderIncrement_highContention() { @Benchmark @Threads(1) public void accumulatorIncrement_lowContention() { - Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); + accumulator.inc(Counter.HITS); } @Benchmark @Threads(Threads.MAX) public void accumulatorIncrement_highContention() { - Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); - } - - /** - * The typed {@link Accumulator} wrapper's {@link Accumulator#inc}, paired against {@code - * accumulatorIncrement*} above: same underlying {@link Accumulator.EmbeddingSupport#inc} call, - * one extra field-load indirection through the instance. Should track the raw numbers closely -- - * a divergence here would mean the indirection isn't being inlined away. - */ - @Benchmark - @Threads(1) - public void typedIncrement_lowContention() { - typedAccumulator.inc(Counter.HITS); - } - - @Benchmark - @Threads(Threads.MAX) - public void typedIncrement_highContention() { - typedAccumulator.inc(Counter.HITS); + accumulator.inc(Counter.HITS); } @Benchmark @@ -232,9 +155,8 @@ public void longAdderSumThenReset_highContention(Blackhole blackhole) { @Benchmark @Threads(1) public void accumulatorAccumulateAndReset_lowContention(Blackhole blackhole) { - Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); - blackhole.consume( - Accumulator.EmbeddingSupport.accumulateAndReset(accumulator, Counter.values().length)); + accumulator.inc(Counter.HITS); + blackhole.consume(accumulator.accumulateAndReset()); } /** @@ -242,80 +164,36 @@ public void accumulatorAccumulateAndReset_lowContention(Blackhole blackhole) { * Threads.MAX} threads are all draining concurrently. Real callers don't do this -- see {@code * accumulatorMixed-write}/{@code accumulatorMixed-drain} below for the "many writers, one rare * drainer" shape this class actually targets. Kept as the worst-case upper bound: no production - * topology should be more contended on {@link Accumulator.EmbeddingSupport#accumulateAndReset} - * than this. + * topology should be more contended on {@link Accumulator#accumulateAndReset} than this. */ @Benchmark @Threads(Threads.MAX) public void accumulatorAccumulateAndReset_highContention(Blackhole blackhole) { - Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); - blackhole.consume( - Accumulator.EmbeddingSupport.accumulateAndReset(accumulator, Counter.values().length)); + accumulator.inc(Counter.HITS); + blackhole.consume(accumulator.accumulateAndReset()); } /** * The realistic counterpart to {@code accumulatorAccumulateAndReset_highContention}: many writer * threads incrementing, and a single dedicated thread polling {@link - * Accumulator.EmbeddingSupport#accumulateAndReset} -- not every thread doing both on every op. - * {@code accumulatorMixed-write} measures increment cost while a drain is actively contending for - * stripe locks; {@code accumulatorMixed-drain} measures the drain's own cost under that same live - * write pressure. The 4:1 writer:drainer ratio is illustrative of "many writers, rare drain," not - * tuned to a specific core count. + * Accumulator#accumulateAndReset} -- not every thread doing both on every op. {@code + * accumulatorMixed-write} measures increment cost while a drain is actively running; {@code + * accumulatorMixed-drain} measures the drain's own cost under that same live write pressure. The + * 4:1 writer:drainer ratio is illustrative of "many writers, rare drain," not tuned to a specific + * core count. */ @Benchmark @Group("accumulatorMixed") @GroupThreads(4) public void accumulatorMixed_write() { - Accumulator.EmbeddingSupport.inc(accumulator, Counter.HITS); + accumulator.inc(Counter.HITS); } @Benchmark @Group("accumulatorMixed") @GroupThreads(1) public void accumulatorMixed_drain(Blackhole blackhole) { - blackhole.consume( - Accumulator.EmbeddingSupport.accumulateAndReset(accumulator, Counter.values().length)); - } - - /** - * {@link Accumulator#update}, paired against the raw {@link Accumulator.EmbeddingSupport#update} - * lock/dispatch it wraps: the mutator here constructs a {@link Accumulator.Stripe} under the held - * lock and immediately lets it go, which is exactly the "small, non-capturing, non-escaping" - * shape documented as a scalar-replacement candidate. If escape analysis is doing its job, this - * tracks the raw call closely; if it regresses (e.g. after a JIT/JDK change, or a mutator shape - * that stops inlining), this is the number that would move. - */ - @Benchmark - @Threads(1) - public void typedUpdate_lowContention() { - typedAccumulator.update(stripe -> stripe.inc(Counter.HITS)); - } - - @Benchmark - @Threads(Threads.MAX) - public void typedUpdate_highContention() { - typedAccumulator.update(stripe -> stripe.inc(Counter.HITS)); - } - - /** - * {@link Accumulator#accumulateAndReset}, paired against the raw {@link - * Accumulator.EmbeddingSupport#accumulateAndReset} it wraps. Unlike {@link Accumulator.Stripe}, - * {@link Accumulator.Counts} is documented to escape (the caller holds and reads it after - * return), so this is expected to run measurably slower than the raw call by roughly one small - * object allocation per drain -- not a scalar-replacement candidate, and not meant to look free. - */ - @Benchmark - @Threads(1) - public void typedAccumulateAndReset_lowContention(Blackhole blackhole) { - typedAccumulator.inc(Counter.HITS); - blackhole.consume(typedAccumulator.accumulateAndReset()); - } - - @Benchmark - @Threads(Threads.MAX) - public void typedAccumulateAndReset_highContention(Blackhole blackhole) { - typedAccumulator.inc(Counter.HITS); - blackhole.consume(typedAccumulator.accumulateAndReset()); + blackhole.consume(accumulator.accumulateAndReset()); } @Benchmark diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index 746c985a5b6..bb830b4acce 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -1,44 +1,47 @@ package datadog.trace.util; import datadog.environment.ThreadSupport; -import datadog.trace.api.function.Strategy; -import datadog.trace.api.function.StrategyConsumer; -import java.util.Arrays; -import java.util.function.BiConsumer; -import java.util.function.Consumer; -import java.util.function.ObjLongConsumer; -import javax.annotation.ParametersAreNonnullByDefault; -import javax.annotation.concurrent.GuardedBy; +import java.util.concurrent.atomic.AtomicLongArray; /** - * A typed, instance-owning wrapper over {@link EmbeddingSupport}: ties an enum's type to its - * backing {@code long[][]} at construction, so {@link #inc}/{@link #add} can't be called with a key - * from a different enum than the one this accumulator was {@link #of created} for. Costs one - * field-load indirection per call versus calling {@link EmbeddingSupport} directly -- the same - * trade {@code StringIndex} makes over its own nested {@code EmbeddingSupport}. + * A striped, lock-free counter primitive keyed by enum ordinal: {@code LongAdder}'s write + * scalability, but as one shared, thread-sharded table instead of one independent {@code LongAdder} + * per counter -- which avoids paying {@code LongAdder}'s per-instance striping overhead {@code + * E.values().length} times over. * *

{@code
  * enum MyCounters { FOO, BAR }
  *
  * Accumulator counters = Accumulator.of(MyCounters.values());
  * counters.inc(MyCounters.FOO);
- * counters.update(stripe -> {
- *   stripe.inc(MyCounters.FOO);
- *   stripe.inc(MyCounters.BAR);
- * });
+ * counters.add(MyCounters.BAR, 5L);
  *
- * Accumulator.Counts drained = counters.accumulateAndReset(); // atomically per stripe
+ * Accumulator.Counts drained = counters.accumulateAndReset();
  * long foo = drained.get(MyCounters.FOO);
  * }
* - * @see EmbeddingSupport + *

Each counter's own {@link #accumulateAndReset} slot is read-and-zeroed with a single atomic + * {@code getAndSet}, so -- like {@code Accumulator}'s previous {@code synchronized}-stripe design, + * and unlike {@code LongAdder#sumThenReset()} -- no individual increment can land in the gap + * between summing and zeroing and be silently lost. What's gone is the previous design's + * row-wide atomicity: {@link #inc}/{@link #add} for two different counters are no longer + * guaranteed to be seen together by a concurrent {@link #accumulateAndReset}. There is no {@code + * update}-style escape hatch for grouping several counters under one atomic operation -- callers + * needing that must weigh whether the guarantee was load-bearing (most call sites are logging + * unrelated aspects of the same event, not maintaining a cross-counter invariant a reader depends + * on) or bring their own coordination. */ public final class Accumulator> { - private final long[][] data; + /** One full cache line of {@code long}s (64 bytes), used to pad each stripe row. */ + private static final int CACHE_LINE_LONGS = 8; + + private final AtomicLongArray[] data; + private final int width; private final E[] values; - private Accumulator(long[][] data, E[] values) { + private Accumulator(AtomicLongArray[] data, int width, E[] values) { this.data = data; + this.width = width; this.values = values; } @@ -46,7 +49,14 @@ private Accumulator(long[][] data, E[] values) { * @param values the enum constants naming each counter, e.g. {@code MyCounters.values()} */ public static > Accumulator of(E[] values) { - return new Accumulator<>(EmbeddingSupport.create(values), values); + int width = values.length; + int paddedWidth = paddedWidth(width); + int stripes = stripeCount(); + AtomicLongArray[] data = new AtomicLongArray[stripes]; + for (int i = 0; i < stripes; i++) { + data[i] = new AtomicLongArray(paddedWidth); + } + return new Accumulator<>(data, width, values); } /** @@ -58,118 +68,29 @@ public static > Accumulator of(Class enumType) { /** Increments the counter named by {@code key} in the calling thread's stripe by one. */ public void inc(E key) { - EmbeddingSupport.inc(data, key); + add(key, 1L); } /** Adds {@code delta} to the counter named by {@code key} in the calling thread's stripe. */ public void add(E key, long delta) { - EmbeddingSupport.add(data, key, delta); - } - - /** - * Runs {@code mutator} against a typed view of the calling thread's stripe under a single held - * lock -- the escape hatch for performing several related updates atomically with respect to a - * concurrent {@link #accumulateAndReset}. - * - * @param mutator a strategy over the selected stripe; keep it small and non-capturing so it - * inlines into the lock's critical section, and don't let the {@link Stripe} escape it (store - * it, return it, hand it to another thread) -- see {@link Stripe} - */ - @StrategyConsumer - public void update(@Strategy Consumer> mutator) { - long[] stripe = EmbeddingSupport.stripeOf(data); - synchronized (stripe) { - mutator.accept(new Stripe<>(stripe)); - } - } - - /** - * Like {@link #update(Consumer)}, but passes {@code context} to {@code mutator} as an explicit - * parameter instead of letting the mutator capture it -- for a caller that would otherwise need - * to close over a local (e.g. a count) just to get it into the critical section. Note {@code - * context} is boxed if it's a primitive at the call site; that's a real allocation trade against - * the capturing lambda it replaces, not a free win -- prefer this only when {@code context} would - * otherwise be the only thing forcing a capture. For an {@code int} or {@code long} context, use - * {@link #update(long, ObjLongConsumer)} instead to avoid that boxing entirely. - * - * @param context a value the mutator needs, passed in rather than captured - * @param mutator a strategy over {@code context} and the selected stripe; keep it small and - * non-capturing so it inlines into the lock's critical section, and don't let the {@link - * Stripe} escape it (store it, return it, hand it to another thread) -- see {@link Stripe} - */ - @StrategyConsumer - public void update(C context, @Strategy BiConsumer> mutator) { - long[] stripe = EmbeddingSupport.stripeOf(data); - synchronized (stripe) { - mutator.accept(context, new Stripe<>(stripe)); - } - } - - /** - * Like {@link #update(Object, BiConsumer)}, but for a {@code long} context -- reuses the JDK's - * {@link ObjLongConsumer} instead of the generic {@link BiConsumer}, so {@code context} is passed - * as a primitive {@code long} rather than boxed into a {@link Long}. Covers an {@code int} - * context too: it widens to {@code long} for free at the call site, no boxing either way. (A - * dedicated {@code int} overload isn't offered alongside this one -- an {@code int} argument - * would be ambiguous between the two, since it's an exact match for one and a free widening - * conversion to the other, and unrelated functional-interface types block the usual most-specific - * tiebreak.) - * - *

Note the parameter order this forces: {@link ObjLongConsumer#accept} takes {@code (T, - * long)}, so the mutator sees the stripe first and the context second -- the opposite order from - * {@link #update(Object, BiConsumer)}. - * - * @param context a primitive value the mutator needs, passed in rather than captured or boxed - * @param mutator a strategy over the selected stripe and {@code context}; keep it small and - * non-capturing so it inlines into the lock's critical section, and don't let the {@link - * Stripe} escape it (store it, return it, hand it to another thread) -- see {@link Stripe} - */ - @StrategyConsumer - public void update(long context, @Strategy ObjLongConsumer> mutator) { - long[] stripe = EmbeddingSupport.stripeOf(data); - synchronized (stripe) { - mutator.accept(new Stripe<>(stripe), context); - } + stripeOf(data).getAndAdd(key.ordinal(), delta); } /** - * A typed view over one stripe, handed to an {@link #update} strategy: the same enum-ordinal type - * checking {@link Accumulator} provides at the top level, applied inside the critical section - * too. - * - *

Constructed fresh under the held lock on every {@link #update} call. A well-behaved {@link - * Strategy} mutator -- small, non-capturing, and never storing or returning this object -- lets - * escape analysis prove it doesn't escape the inlined call and scalar-replace it, so no - * allocation survives to run time. Break those rules (capture it in a field, return it, hand it - * to another thread) and it degrades to a real, per-call allocation instead of a compile-time - * fiction with no correctness difference either way -- just a cost one. - */ - public static final class Stripe> { - private final long[] stripe; - - private Stripe(long[] stripe) { - this.stripe = stripe; - } - - /** Increments the counter named by {@code key} in this stripe by one. */ - public void inc(E key) { - EmbeddingSupport.inc(stripe, key); - } - - /** Adds {@code delta} to the counter named by {@code key} in this stripe. */ - public void add(E key, long delta) { - EmbeddingSupport.add(stripe, key, delta); - } - } - - /** - * Combines and resets every stripe, returning the sum as a typed view. + * Combines and resets every stripe, returning the sum as a typed view. Each counter is + * read-and-zeroed with one atomic {@code getAndSet} -- see the class-level note on what atomicity + * this does and doesn't provide across different counters. * * @return the sum, keyed by the enum's {@code ordinal()} - * @see EmbeddingSupport#accumulateAndReset */ public Counts accumulateAndReset() { - return new Counts<>(EmbeddingSupport.accumulateAndReset(data, values.length), values); + long[] acc = new long[width]; + for (AtomicLongArray stripe : data) { + for (int i = 0; i < width; i++) { + acc[i] += stripe.getAndSet(i, 0L); + } + } + return new Counts<>(acc, values); } /** @@ -178,21 +99,21 @@ public Counts accumulateAndReset() { * the delta a concurrent {@link #accumulateAndReset} on a reporting cadence is about to report. * * @return the sum, keyed by the enum's {@code ordinal()} - * @see EmbeddingSupport#sum(long[][], int) */ public Counts sum() { - return new Counts<>(EmbeddingSupport.sum(data, values.length), values); + long[] acc = new long[width]; + for (AtomicLongArray stripe : data) { + for (int i = 0; i < width; i++) { + acc[i] += stripe.get(i); + } + } + return new Counts<>(acc, values); } /** - * A typed view over a drained {@code long[]}, returned by {@link #accumulateAndReset} or {@link - * #sum}: the same enum-ordinal type checking {@link Accumulator} provides on writes, applied to - * the read side too. - * - *

Unlike {@link Stripe}, this is expected to escape -- the caller holds and reads it after the - * call returns -- so it's a real, per-drain allocation, not a scalar-replacement candidate. - * That's fine: {@link #accumulateAndReset} runs on a reporting cadence, not per {@link - * #inc}/{@link #add} call. + * A typed view over a drained snapshot, returned by {@link #accumulateAndReset} or {@link #sum}: + * the same enum-ordinal type checking {@link Accumulator} provides on writes, applied to the read + * side too. */ public static final class Counts> { private final long[] counts; @@ -251,232 +172,39 @@ public Counts plus(Counts other) { } /** - * The static, raw-array tier of the striped accumulator primitive: {@code LongAdder}'s write - * scalability, without {@code LongAdder}'s reset hazard. + * The calling thread's stripe: cheap masking, no allocation, no map lookup. * - *

{@code LongAdder#sumThenReset()} is documented as not atomic against concurrent - * updates: an increment landing on a cell after it's summed but before it's zeroed is silently - * and permanently lost. {@link #accumulateAndReset} closes that window by combining and resetting - * each stripe under the same lock that guards its writers. - * - *

Each stripe's state is a bare {@code long[]}, not a named-field struct. An {@code enum} - * assigns a name to each position via its ordinal, so name and position are the same declaration - * and cannot drift apart. This also makes {@link #combine} and {@link #reset} generic, - * branchless, fixed-trip-count array loops -- the shape designed to take advantage of SIMD / - * vector operations on modern hardware -- so they are implemented once here instead of once per - * caller. - * - *

This is a pure namespace over caller-owned {@code long[][]} state -- it allocates no - * container object and is not itself a strategy consumer's receiver. That means {@code create}'s - * type parameter is not bound to the one later {@code inc}/{@code add} calls infer: nothing stops - * a caller from indexing the same {@code long[][]} with a different enum than the one it was - * {@link #create}d for, which silently reads/writes the wrong slot rather than failing to - * compile. Prefer the owning {@link Accumulator} instance, which closes that hole for one - * field-load indirection per call; reach for this class directly only when that indirection is - * worth removing. - * - *

{@code
-   * enum MyCounters { FOO, BAR }
-   *
-   * long[][] data = Accumulator.EmbeddingSupport.create(MyCounters.values());
-   * Accumulator.EmbeddingSupport.inc(data, MyCounters.FOO);
-   * Accumulator.EmbeddingSupport.update(data, stripe -> {
-   *   Accumulator.EmbeddingSupport.inc(stripe, MyCounters.FOO);
-   *   Accumulator.EmbeddingSupport.inc(stripe, MyCounters.BAR);
-   * });
-   *
-   * long[] drained =
-   *     Accumulator.EmbeddingSupport.accumulateAndReset(data, MyCounters.values().length);
-   * long foo = drained[MyCounters.FOO.ordinal()];
-   * }
+ *

Multiple threads can map to the same stripe (this is masking, not a bijection); each + * counter's own atomic slot makes that safe, just not maximally scalable under a hash collision. */ - @ParametersAreNonnullByDefault - public static final class EmbeddingSupport { - private EmbeddingSupport() {} - - /** One full cache line of {@code long}s (64 bytes), used to pad each stripe's row. */ - private static final int CACHE_LINE_LONGS = 8; - - /** - * Creates the backing storage for an accumulator over {@code values}: one {@code long[]} row - * per stripe, sized to {@code values.length} plus at least one trailing cache line of padding - * so adjacent stripe rows don't false-share. - * - *

Stripe count is fixed at a power of two oversized to roughly 2x {@link - * Runtime#availableProcessors()} (minimum 4); it is not a per-call knob (see {@link - * #stripeCount()}). - * - * @param values the enum constants naming each counter, e.g. {@code MyCounters.values()} - * @return a new {@code long[stripeCount][paddedWidth]} array, zero-initialized - */ - public static > long[][] create(E[] values) { - int paddedWidth = paddedWidth(values.length); - int stripes = stripeCount(); - long[][] data = new long[stripes][]; - for (int i = 0; i < stripes; i++) { - data[i] = new long[paddedWidth]; - } - return data; - } - - /** - * Increments the counter named by {@code key} in the calling thread's stripe by one. - * - *

Convenience for the common case: selects the calling thread's stripe, takes its lock, and - * increments. To perform several increments under a single held lock, use {@link #update}. - */ - public static > void inc(long[][] data, E key) { - add(data, key, 1L); - } - - /** - * Adds {@code delta} to the counter named by {@code key} in the calling thread's stripe. - * - * @see #inc(long[][], Enum) - */ - public static > void add(long[][] data, E key, long delta) { - add(stripeOf(data), key, delta); - } - - /** - * Increments the counter named by {@code key} in {@code stripe} by one, under {@code stripe}'s - * own lock. - * - *

Intended for use inside an {@link #update} lambda, where {@code stripe} is already the - * calling thread's selected row: {@code synchronized} is reentrant, so calling this here does - * not deadlock or take a second lock. - */ - public static > void inc(long[] stripe, E key) { - add(stripe, key, 1L); - } - - /** - * Adds {@code delta} to the counter named by {@code key} in {@code stripe}, under {@code - * stripe}'s own lock. - * - * @see #inc(long[], Enum) - */ - public static > void add(long[] stripe, E key, long delta) { - synchronized (stripe) { - stripe[key.ordinal()] += delta; - } - } - - /** - * Runs {@code mutator} against the calling thread's stripe under a single held lock -- the - * escape hatch for performing several related updates atomically with respect to a concurrent - * {@link #accumulateAndReset}. - * - * @param mutator a strategy over the selected stripe; keep it small and non-capturing so it - * inlines into the lock's critical section - */ - @StrategyConsumer - public static void update(long[][] data, @Strategy Consumer mutator) { - long[] stripe = stripeOf(data); - synchronized (stripe) { - mutator.accept(stripe); - } - } - - /** - * Combines and resets every stripe, returning the sum. Each stripe is locked for exactly as - * long as it takes to fold its values into the result and zero it -- the same lock held by - * {@link #inc}/{@link #add}/{@link #update} -- so no writer can land an increment in the gap - * between summing and zeroing the way {@code LongAdder#sumThenReset()} allows. - * - *

Only the first {@code width} positions of each stripe are read or written -- the trailing - * cache line {@link #paddedWidth} reserves past that point is never touched again after {@link - * #create} zero-initializes it, so it stays a genuinely dead buffer between adjacent stripe - * rows instead of being read-and-rewritten (dirtying that cache line) on every drain. - * - * @param width the number of counters actually in use, e.g. {@code values.length} - * @return a new array of length {@code width}, indexed by the enum's {@code ordinal()} - */ - public static long[] accumulateAndReset(long[][] data, int width) { - long[] acc = new long[width]; - for (long[] stripe : data) { - synchronized (stripe) { - combine(acc, stripe, width); - reset(stripe, width); - } - } - return acc; - } - - /** - * Combines every stripe without resetting it, returning the sum -- a live, non-destructive - * snapshot for a diagnostic read that must not perturb the delta a concurrent {@link - * #accumulateAndReset} on a reporting cadence is about to report. - * - * @param width the number of counters actually in use, e.g. {@code values.length} - * @return a new array of length {@code width}, indexed by the enum's {@code ordinal()} - * @see #accumulateAndReset(long[][], int) - */ - public static long[] sum(long[][] data, int width) { - long[] acc = new long[width]; - for (long[] stripe : data) { - synchronized (stripe) { - combine(acc, stripe, width); - } - } - return acc; - } - - /** - * {@code acc[i] += stripe[i]} for {@code i} in {@code [0, width)} -- a fixed-trip-count loop C2 - * can auto-vectorize. - */ - @GuardedBy("stripe") - private static void combine(long[] acc, long[] stripe, int width) { - for (int i = 0; i < width; i++) { - acc[i] += stripe[i]; - } - } - - /** - * Zeroes {@code stripe}'s first {@code width} positions, via the JVM-intrinsic {@link - * Arrays#fill}. Deliberately stops at {@code width}, leaving the trailing padding untouched. - */ - @GuardedBy("stripe") - private static void reset(long[] stripe, int width) { - Arrays.fill(stripe, 0, width, 0L); - } - - /** - * The calling thread's stripe: cheap masking, no allocation, no map lookup. - * - *

Multiple threads can map to the same stripe (this is masking, not a bijection); each - * stripe's own lock makes that safe, just not maximally scalable under a hash collision. - */ - private static long[] stripeOf(long[][] data) { - int mask = data.length - 1; - int idx = (int) (ThreadSupport.threadId() & mask); - return data[idx]; - } + private static AtomicLongArray stripeOf(AtomicLongArray[] data) { + int mask = data.length - 1; + int idx = (int) (ThreadSupport.threadId() & mask); + return data[idx]; + } - /** - * A fixed, power-of-two stripe count deliberately oversized to roughly 2x {@link - * Runtime#availableProcessors()} (minimum 4). Not exposed as a per-call override: a mandatory - * sizing knob on every caller fails the "print test" of self-explanatory API design. - * - *

Sizing to exactly the core count leaves stripe collisions likely under real contention - * (birthday-paradox math: with {@code n} contending threads and {@code m} stripes, expected - * colliding pairs are {@code n(n-1)/(2m)}) -- and a collision costs a blocking {@code - * synchronized} wait, not a cheap CAS retry. Doubling the stripe count roughly halves that - * collision count for a one-time, per-accumulator memory cost, at the price of a slightly more - * expensive (but far rarer) {@link #accumulateAndReset} drain -- the right trade given {@link - * #inc}/ {@link #add} run on every call while {@link #accumulateAndReset} runs on a reporting - * cadence. - */ - private static int stripeCount() { - int cpus = Runtime.getRuntime().availableProcessors(); - return Math.max(4, 2 * Integer.highestOneBit(Math.max(1, cpus))); - } + /** + * A fixed, power-of-two stripe count deliberately oversized to roughly 2x {@link + * Runtime#availableProcessors()} (minimum 4). Not exposed as a per-call override: a mandatory + * sizing knob on every caller fails the "print test" of self-explanatory API design. + * + *

Sizing to exactly the core count leaves stripe collisions likely under real contention + * (birthday-paradox math: with {@code n} contending threads and {@code m} stripes, expected + * colliding pairs are {@code n(n-1)/(2m)}) -- and a collision costs CAS-retry/cache-line-bounce + * cost, the same problem {@code LongAdder}'s own {@code Cell[]} table exists to avoid. Doubling + * the stripe count roughly halves that collision count for a one-time, per-accumulator memory + * cost, at the price of a slightly more expensive (but far rarer) {@link #accumulateAndReset} + * drain -- the right trade given {@link #inc}/{@link #add} run on every call while {@link + * #accumulateAndReset} runs on a reporting cadence. + */ + private static int stripeCount() { + int cpus = Runtime.getRuntime().availableProcessors(); + return Math.max(4, 2 * Integer.highestOneBit(Math.max(1, cpus))); + } - /** Rounds {@code width} up to a whole number of cache lines, plus one full trailing line. */ - private static int paddedWidth(int width) { - int wholeLines = ((width + CACHE_LINE_LONGS - 1) / CACHE_LINE_LONGS) * CACHE_LINE_LONGS; - return wholeLines + CACHE_LINE_LONGS; - } + /** Rounds {@code width} up to a whole number of cache lines, plus one full trailing line. */ + private static int paddedWidth(int width) { + int wholeLines = ((width + CACHE_LINE_LONGS - 1) / CACHE_LINE_LONGS) * CACHE_LINE_LONGS; + return wholeLines + CACHE_LINE_LONGS; } } diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java index 868f2ab6d0d..a8f2f6f39d8 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java @@ -27,24 +27,14 @@ * what each actually costs once the counters they represent are hit by real concurrent writers, as * they are on the telemetry paths this class targets. * - *

Measured on a 10-CPU machine (JDK 1.8.0_382 Zulu), 4 counters, {@code - * Accumulator.EmbeddingSupport.stripeCount()} = 16: - * - *

{@code
- * fresh:      4 LongAdders =    160 bytes, Accumulator = 2384 bytes
- * contended:  4 LongAdders =  17560 bytes, Accumulator = 2384 bytes
- * }
- * - * Finding: fresh, {@code LongAdder} looks ~15x lighter -- but that's an artifact of never having - * been written to concurrently. Once real contention forces each {@code LongAdder}'s {@code Cell[]} - * table to grow (each {@code Cell} is {@code @Contended}-padded against false sharing, the same - * problem {@link Accumulator}'s own padding solves), the four {@code LongAdder}s alone end up over - * 7x heavier than {@code Accumulator}'s entire fixed footprint -- and {@code Accumulator} does not - * grow further as more contention arrives within its existing stripe count, while every additional - * concurrently-written {@code LongAdder} keeps paying this cost independently. {@code - * Accumulator}'s up-front cost is the more predictable one: fixed at creation, independent of - * runtime contention, and shared (one striped array) across however many counters the caller's enum - * declares, rather than paid per counter. + *

{@link Accumulator}'s stripe count is fixed at creation (roughly 2x {@link + * Runtime#availableProcessors()}, minimum 4) and does not grow further as more contention arrives + * within it, while every additional concurrently-written {@code LongAdder} keeps paying its own + * {@code Cell[]} growth cost independently. {@code Accumulator}'s up-front cost is the more + * predictable one: fixed at creation, independent of runtime contention, and shared (one striped + * table) across however many counters the caller's enum declares, rather than paid per counter. The + * printed numbers below vary by run/JVM -- see the assertions for the invariants that actually + * matter. */ class AccumulatorFootprintTest { @@ -78,7 +68,7 @@ static LongAdder[] freshAdders() { @Test void freshFootprint() { LongAdder[] adders = freshAdders(); - long[][] accumulator = Accumulator.EmbeddingSupport.create(Counters.values()); + Accumulator accumulator = Accumulator.of(Counters.values()); long adderBytes = bytes((Object) adders); long accumulatorBytes = bytes(accumulator); @@ -133,7 +123,7 @@ void contendedFootprint() throws InterruptedException { } long contendedAdderBytes = bytes((Object) adders); - long[][] accumulator = Accumulator.EmbeddingSupport.create(Counters.values()); + Accumulator accumulator = Accumulator.of(Counters.values()); long accumulatorBytes = bytes(accumulator); System.out.printf( diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java index a767b803476..f17eab823fb 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -23,82 +23,53 @@ enum Counters { @Test void freshAccumulatorSumsToZero() { - long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); - long[] drained = - Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); + Accumulator counters = Accumulator.of(Counters.values()); + Accumulator.Counts drained = counters.accumulateAndReset(); for (Counters c : Counters.values()) { - assertEquals(0L, drained[c.ordinal()]); + assertEquals(0L, drained.get(c)); } } @Test void incIncrementsByOne() { - long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); - Accumulator.EmbeddingSupport.inc(data, Counters.FOO); - Accumulator.EmbeddingSupport.inc(data, Counters.FOO); - Accumulator.EmbeddingSupport.inc(data, Counters.BAR); - - long[] drained = - Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); - assertEquals(2L, drained[Counters.FOO.ordinal()]); - assertEquals(1L, drained[Counters.BAR.ordinal()]); - assertEquals(0L, drained[Counters.BAZ.ordinal()]); + Accumulator counters = Accumulator.of(Counters.values()); + counters.inc(Counters.FOO); + counters.inc(Counters.FOO); + counters.inc(Counters.BAR); + + Accumulator.Counts drained = counters.accumulateAndReset(); + assertEquals(2L, drained.get(Counters.FOO)); + assertEquals(1L, drained.get(Counters.BAR)); + assertEquals(0L, drained.get(Counters.BAZ)); } @Test void addAppliesArbitraryDelta() { - long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); - Accumulator.EmbeddingSupport.add(data, Counters.BAZ, 41L); - Accumulator.EmbeddingSupport.add(data, Counters.BAZ, 1L); - - long[] drained = - Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); - assertEquals(42L, drained[Counters.BAZ.ordinal()]); - } + Accumulator counters = Accumulator.of(Counters.values()); + counters.add(Counters.BAZ, 41L); + counters.add(Counters.BAZ, 1L); - @Test - void updateAppliesSeveralOpsUnderOneLock() { - long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); - Accumulator.EmbeddingSupport.update( - data, - stripe -> { - Accumulator.EmbeddingSupport.inc(stripe, Counters.FOO); - Accumulator.EmbeddingSupport.inc(stripe, Counters.FOO); - Accumulator.EmbeddingSupport.add(stripe, Counters.BAR, 5L); - }); - - long[] drained = - Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); - assertEquals(2L, drained[Counters.FOO.ordinal()]); - assertEquals(5L, drained[Counters.BAR.ordinal()]); + Accumulator.Counts drained = counters.accumulateAndReset(); + assertEquals(42L, drained.get(Counters.BAZ)); } @Test void accumulateAndResetsSoASecondDrainIsZero() { - long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); - Accumulator.EmbeddingSupport.inc(data, Counters.FOO); + Accumulator counters = Accumulator.of(Counters.values()); + counters.inc(Counters.FOO); - long[] first = Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); - assertEquals(1L, first[Counters.FOO.ordinal()]); + Accumulator.Counts first = counters.accumulateAndReset(); + assertEquals(1L, first.get(Counters.FOO)); - long[] second = Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); + Accumulator.Counts second = counters.accumulateAndReset(); for (Counters c : Counters.values()) { - assertEquals(0L, second[c.ordinal()]); + assertEquals(0L, second.get(c)); } } - @Test - void drainedArrayIsExactlyWidthLong() { - long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); - long[] drained = - Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); - assertEquals(Counters.values().length, drained.length); - assertTrue(drained.length < data[0].length, "drained array should exclude stripe padding"); - } - @Test void concurrentIncrementsAreNotLost() throws InterruptedException { - long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); + Accumulator counters = Accumulator.of(Counters.values()); int threadCount = 16; int incrementsPerThread = 10_000; @@ -112,7 +83,7 @@ void concurrentIncrementsAreNotLost() throws InterruptedException { try { start.await(); for (int i = 0; i < incrementsPerThread; i++) { - Accumulator.EmbeddingSupport.inc(data, Counters.FOO); + counters.inc(Counters.FOO); } } catch (InterruptedException e) { Thread.currentThread().interrupt(); @@ -127,15 +98,14 @@ void concurrentIncrementsAreNotLost() throws InterruptedException { pool.shutdown(); } - long[] drained = - Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); - assertEquals((long) threadCount * incrementsPerThread, drained[Counters.FOO.ordinal()]); + Accumulator.Counts drained = counters.accumulateAndReset(); + assertEquals((long) threadCount * incrementsPerThread, drained.get(Counters.FOO)); } @Test void concurrentAccumulateAndDuringWritesNeverExceedsWritten() throws InterruptedException, ExecutionException, TimeoutException { - long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); + Accumulator counters = Accumulator.of(Counters.values()); int threadCount = 8; int incrementsPerThread = 5_000; @@ -149,11 +119,9 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten() pool.submit( () -> { while (!stop.get()) { - long[] drained = - Accumulator.EmbeddingSupport.accumulateAndReset( - data, Counters.values().length); + Accumulator.Counts drained = counters.accumulateAndReset(); synchronized (runningTotal) { - runningTotal[0] += drained[Counters.FOO.ordinal()]; + runningTotal[0] += drained.get(Counters.FOO); } } }); @@ -162,7 +130,7 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten() pool.execute( () -> { for (int i = 0; i < incrementsPerThread; i++) { - Accumulator.EmbeddingSupport.inc(data, Counters.FOO); + counters.inc(Counters.FOO); } done.countDown(); }); @@ -172,10 +140,9 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten() stop.set(true); drainer.get(30, TimeUnit.SECONDS); - long[] finalDrain = - Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); + Accumulator.Counts finalDrain = counters.accumulateAndReset(); synchronized (runningTotal) { - runningTotal[0] += finalDrain[Counters.FOO.ordinal()]; + runningTotal[0] += finalDrain.get(Counters.FOO); } assertEquals((long) threadCount * incrementsPerThread, runningTotal[0]); @@ -186,47 +153,30 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten() @Test void sumDoesNotResetStripes() { - long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); - Accumulator.EmbeddingSupport.inc(data, Counters.FOO); + Accumulator counters = Accumulator.of(Counters.values()); + counters.inc(Counters.FOO); - long[] first = Accumulator.EmbeddingSupport.sum(data, Counters.values().length); - assertEquals(1L, first[Counters.FOO.ordinal()]); + Accumulator.Counts first = counters.sum(); + assertEquals(1L, first.get(Counters.FOO)); // sum() didn't reset anything, so a second sum() sees the same total - long[] second = Accumulator.EmbeddingSupport.sum(data, Counters.values().length); - assertEquals(1L, second[Counters.FOO.ordinal()]); + Accumulator.Counts second = counters.sum(); + assertEquals(1L, second.get(Counters.FOO)); // and a real drain afterwards still sees the value sum() didn't consume - long[] drained = - Accumulator.EmbeddingSupport.accumulateAndReset(data, Counters.values().length); - assertEquals(1L, drained[Counters.FOO.ordinal()]); + Accumulator.Counts drained = counters.accumulateAndReset(); + assertEquals(1L, drained.get(Counters.FOO)); } @Test void sumReflectsIncrementsMadeAfterAnEarlierSum() { - long[][] data = Accumulator.EmbeddingSupport.create(Counters.values()); - Accumulator.EmbeddingSupport.inc(data, Counters.FOO); - Accumulator.EmbeddingSupport.sum(data, Counters.values().length); - - Accumulator.EmbeddingSupport.inc(data, Counters.FOO); - long[] second = Accumulator.EmbeddingSupport.sum(data, Counters.values().length); - assertEquals(2L, second[Counters.FOO.ordinal()]); - } - - @Test - void typedWrapperSumDoesNotReset() { Accumulator counters = Accumulator.of(Counters.values()); counters.inc(Counters.FOO); - counters.add(Counters.BAR, 5L); + counters.sum(); - Accumulator.Counts sum = counters.sum(); - assertEquals(1L, sum.get(Counters.FOO)); - assertEquals(5L, sum.get(Counters.BAR)); - - // still there for the real drain - Accumulator.Counts drained = counters.accumulateAndReset(); - assertEquals(1L, drained.get(Counters.FOO)); - assertEquals(5L, drained.get(Counters.BAR)); + counters.inc(Counters.FOO); + Accumulator.Counts second = counters.sum(); + assertEquals(2L, second.get(Counters.FOO)); } @Test @@ -288,68 +238,4 @@ void plusCombinesAStoredRunningTotalWithALiveSumWithoutMutatingEither() { assertEquals(1L, storedTotal.get(Counters.FOO)); assertEquals(1L, counters.sum().get(Counters.FOO)); } - - @Test - void typedWrapperDelegatesToEmbeddingSupport() { - Accumulator counters = Accumulator.of(Counters.values()); - counters.inc(Counters.FOO); - counters.inc(Counters.FOO); - counters.add(Counters.BAR, 5L); - counters.update(stripe -> stripe.inc(Counters.BAZ)); - - Accumulator.Counts drained = counters.accumulateAndReset(); - assertEquals(2L, drained.get(Counters.FOO)); - assertEquals(5L, drained.get(Counters.BAR)); - assertEquals(1L, drained.get(Counters.BAZ)); - } - - @Test - void contextualUpdatePassesContextInsteadOfCapturingIt() { - Accumulator counters = Accumulator.of(Counters.values()); - String context = "abcde"; - - counters.update( - context, - (ctx, stripe) -> { - stripe.inc(Counters.FOO); - stripe.add(Counters.BAR, ctx.length()); - }); - - Accumulator.Counts drained = counters.accumulateAndReset(); - assertEquals(1L, drained.get(Counters.FOO)); - assertEquals(5L, drained.get(Counters.BAR)); - } - - @Test - void intContextWidensIntoTheLongOverloadWithoutBoxing() { - Accumulator counters = Accumulator.of(Counters.values()); - int delta = 5; - - counters.update( - delta, - (stripe, d) -> { - stripe.inc(Counters.FOO); - stripe.add(Counters.BAR, d); - }); - - Accumulator.Counts drained = counters.accumulateAndReset(); - assertEquals(1L, drained.get(Counters.FOO)); - assertEquals(5L, drained.get(Counters.BAR)); - } - - @Test - void longContextualUpdateAvoidsBoxing() { - Accumulator counters = Accumulator.of(Counters.values()); - - counters.update( - 5L, - (stripe, delta) -> { - stripe.inc(Counters.FOO); - stripe.add(Counters.BAR, delta); - }); - - Accumulator.Counts drained = counters.accumulateAndReset(); - assertEquals(1L, drained.get(Counters.FOO)); - assertEquals(5L, drained.get(Counters.BAR)); - } } From e90422c89e943f0b0382b90a69a6e51cf8a1bac2 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Wed, 2 Sep 2026 18:52:48 -0400 Subject: [PATCH 21/36] Correct AccumulatorBenchmark javadoc's high-contention drain numbers The previous numbers (2.740 us/op) didn't match either measured JMH run on disk (both agree on ~13.4 us/op) -- Accumulator's drain is ~5.5x worse than longAdderGroup at high contention, not "slightly worse". The increment-side win is also corrected (~50x, not ~2 orders of magnitude) though the conclusion there is unchanged. Co-Authored-By: Claude Sonnet 5 --- .../trace/util/AccumulatorBenchmark.java | 28 +++++++++++-------- 1 file changed, 17 insertions(+), 11 deletions(-) diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java index a1e1679c9f0..5bf142edfa8 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -44,18 +44,24 @@ * the write path against {@link Accumulator}'s thread-sharded stripes, especially under contention. * This is the realistic production baseline this class was built to replace (see {@code * TracerHealthMetrics}'s pre-migration design, one {@code LongAdder} field per counter): - * AccumulatorBenchmark.accumulatorIncrement_lowContention avgt 6 0.007 ± 0.001 us/op - * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 6 0.009 ± 0.001 us/op - * AccumulatorBenchmark.longAdderGroupIncrement_lowContention avgt 6 0.012 ± 0.001 us/op - * AccumulatorBenchmark.longAdderGroupIncrement_highContention avgt 6 1.178 ± 0.134 us/op - * AccumulatorBenchmark.accumulatorAccumulateAndReset_lowContention avgt 6 0.054 ± 0.001 us/op - * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 6 2.740 ± 0.210 us/op - * AccumulatorBenchmark.longAdderGroupAccumulateAnd_lowContention avgt 6 0.024 ± 0.001 us/op - * AccumulatorBenchmark.longAdderGroupAccumulateAnd_highContention avgt 6 2.439 ± 0.337 us/op + * AccumulatorBenchmark.accumulatorIncrement_lowContention avgt 6 0.010 ± 0.001 us/op + * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 6 0.025 ± 0.039 us/op + * AccumulatorBenchmark.longAdderGroupIncrement_lowContention avgt 6 0.012 ± 0.001 us/op + * AccumulatorBenchmark.longAdderGroupIncrement_highContention avgt 6 1.178 ± 0.134 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset_lowContention avgt 6 0.104 ± 0.001 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 6 13.357 ± 1.203 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd_lowContention avgt 6 0.024 ± 0.001 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd_highContention avgt 6 2.439 ± 0.337 us/op * At high contention, {@link Accumulator} beats the realistic {@code longAdderGroup} - * baseline by roughly two orders of magnitude on increment (the call that runs on every event) - * while being slightly worse on drain (the call that runs once per reporting cycle) -- a clean win - * once weighted by call-site frequency, not just a wash. + * baseline by nearly 50x on increment (the call that runs on every event), but is itself + * roughly 5.5x worse than {@code longAdderGroup} on drain under that same high-contention + * topology (the call that runs once per reporting cycle) -- {@code accumulateAndReset} walks every + * stripe with a full {@code getAndSet} per counter, so more stripes (sized for core count) means + * more per-drain work than {@code longAdderGroup}'s one-lock-per-counter {@code sumThenReset}. This + * is still a clean win once weighted by call-site frequency -- the increment win is ~50x on a call + * that fires on every event, the drain loss is ~5.5x on a call that fires once per reporting cycle + * (e.g. a 30s flush tick) -- but the drain-side regression is real, not "slightly worse," and worth + * knowing before assuming this trade is free in every topology. */ @State(Scope.Benchmark) @Warmup(iterations = 1, time = 10) From 065ee242f6951592bb7017524bcee0880540f281 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Thu, 3 Sep 2026 07:22:26 -0400 Subject: [PATCH 22/36] Add fair per-thread-distributed 8-wide AccumulatorBenchmark comparison, correct javadoc numbers Width-1 benchmarks pinned every thread to one shared longAdderGroup lock, the degenerate worst case for that baseline. New *8_* benchmarks pin each thread to one of 8 counters instead, giving longAdderGroup a fair shot at distributed locking. Also replaces an earlier javadoc correction that was itself wrong: the 13.357 us/op drain figure it introduced was a correlated anomaly across two low-sample runs, not reproducible at Fork(5)/15 samples. --- .../trace/util/AccumulatorBenchmark.java | 161 +++++++++++++++--- 1 file changed, 137 insertions(+), 24 deletions(-) diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java index 5bf142edfa8..5345f490096 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -3,6 +3,7 @@ import static java.util.concurrent.TimeUnit.MICROSECONDS; import java.util.concurrent.ConcurrentHashMap; +import java.util.concurrent.atomic.AtomicInteger; import java.util.concurrent.atomic.AtomicLong; import java.util.concurrent.atomic.LongAdder; import org.openjdk.jmh.annotations.Benchmark; @@ -39,46 +40,106 @@ * as the payload: one {@code LongAdder} per counter, with a per-counter lock guarding both * the increment and the drain (locking only the drain does nothing -- {@code sumThenReset()}'s * internal race is against the {@code LongAdder}'s own CAS-based {@code add()}, not against any - * lock a caller takes). With this benchmark's single counter, that per-counter lock collapses to - * one lock shared by every thread -- no thread-based distribution at all -- so it loses badly on - * the write path against {@link Accumulator}'s thread-sharded stripes, especially under contention. - * This is the realistic production baseline this class was built to replace (see {@code - * TracerHealthMetrics}'s pre-migration design, one {@code LongAdder} field per counter): - * AccumulatorBenchmark.accumulatorIncrement_lowContention avgt 6 0.010 ± 0.001 us/op - * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 6 0.025 ± 0.039 us/op - * AccumulatorBenchmark.longAdderGroupIncrement_lowContention avgt 6 0.012 ± 0.001 us/op - * AccumulatorBenchmark.longAdderGroupIncrement_highContention avgt 6 1.178 ± 0.134 us/op - * AccumulatorBenchmark.accumulatorAccumulateAndReset_lowContention avgt 6 0.104 ± 0.001 us/op - * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 6 13.357 ± 1.203 us/op - * AccumulatorBenchmark.longAdderGroupAccumulateAnd_lowContention avgt 6 0.024 ± 0.001 us/op - * AccumulatorBenchmark.longAdderGroupAccumulateAnd_highContention avgt 6 2.439 ± 0.337 us/op - * At high contention, {@link Accumulator} beats the realistic {@code longAdderGroup} - * baseline by nearly 50x on increment (the call that runs on every event), but is itself - * roughly 5.5x worse than {@code longAdderGroup} on drain under that same high-contention - * topology (the call that runs once per reporting cycle) -- {@code accumulateAndReset} walks every - * stripe with a full {@code getAndSet} per counter, so more stripes (sized for core count) means - * more per-drain work than {@code longAdderGroup}'s one-lock-per-counter {@code sumThenReset}. This - * is still a clean win once weighted by call-site frequency -- the increment win is ~50x on a call - * that fires on every event, the drain loss is ~5.5x on a call that fires once per reporting cycle - * (e.g. a 30s flush tick) -- but the drain-side regression is real, not "slightly worse," and worth - * knowing before assuming this trade is free in every topology. + * lock a caller takes). The single-counter {@code *_*} benchmarks below collapse that per-counter + * lock to one lock shared by every thread -- the degenerate worst case for {@code longAdderGroup}, + * with no thread-based distribution at all. The {@code *8_*} benchmarks fix that: each JMH worker + * thread is pinned to one of 8 counters for its lifetime (see {@link #threadCounterIndex}), so + * {@code longAdderGroup8}'s threads split into up to 8 groups each contending their own lock -- + * the topology where distributed locking should actually pay off, forcing {@link Accumulator}'s + * thread-striped design to earn its write-side win rather than facing a single-counter worst case. + * Fork(5), 15 samples per benchmark: + * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 15 2.746 ± 0.050 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset_lowContention avgt 15 0.056 ± 0.001 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset8_highContention avgt 15 6.875 ± 0.422 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset8_lowContention avgt 15 0.363 ± 0.003 us/op + * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 15 0.009 ± 0.001 us/op + * AccumulatorBenchmark.accumulatorIncrement_lowContention avgt 15 0.007 ± 0.001 us/op + * AccumulatorBenchmark.accumulatorIncrement8_highContention avgt 15 0.017 ± 0.009 us/op + * AccumulatorBenchmark.accumulatorIncrement8_lowContention avgt 15 0.007 ± 0.001 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd_highContention avgt 15 4.770 ± 1.795 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd_lowContention avgt 15 0.061 ± 0.007 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd8_highContention avgt 15 6.025 ± 0.712 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd8_lowContention avgt 15 0.085 ± 0.004 us/op + * AccumulatorBenchmark.longAdderGroupIncrement_highContention avgt 15 2.294 ± 0.101 us/op + * AccumulatorBenchmark.longAdderGroupIncrement_lowContention avgt 15 0.019 ± 0.001 us/op + * AccumulatorBenchmark.longAdderGroupIncrement8_highContention avgt 15 0.786 ± 0.078 us/op + * AccumulatorBenchmark.longAdderGroupIncrement8_lowContention avgt 15 0.020 ± 0.001 us/op + * On the write side, {@link Accumulator} beats {@code longAdderGroup} at high contention by + * ~255x in the degenerate single-shared-lock case and still by ~46x once counters are fairly spread + * across 8 locks -- a large, reproducible win either way, on the call that runs on every event. + * On the drain side, the two designs are close and the comparison is noisy under contention for + * both: at width 1 {@link Accumulator}'s drain (2.746 us/op) is actually faster than {@code + * longAdderGroup}'s (4.770 ± 1.795 us/op, itself high-variance), and at width 8 it's only ~1.14x + * slower (6.875 vs 6.025 us/op) -- not the regression an earlier reading of this benchmark + * suggested. That earlier reading (13.357 us/op at Fork(2)) turned out to be a correlated anomaly + * across two independent low-sample runs, not a reproducible result; escalating to Fork(5) (15 + * samples) settled it. Net: a large, robust win on the call that fires on every event, and no + * confirmed cost on the call that fires once per reporting cycle. */ @State(Scope.Benchmark) @Warmup(iterations = 1, time = 10) @Measurement(iterations = 3, time = 10) @BenchmarkMode(Mode.AverageTime) @OutputTimeUnit(MICROSECONDS) -@Fork(2) +@Fork(5) public class AccumulatorBenchmark { enum Counter { HITS } + /** + * An 8-constant counterpart to {@link Counter}, used only by the {@code *8_*} benchmarks below. + * Unlike {@link Counter}, where every thread hits the single {@code HITS} constant (the worst + * case for {@code longAdderGroup}'s per-counter locking -- one lock shared by every thread, + * regardless of core count), these benchmarks spread writes across all 8 constants: each JMH + * worker thread is pinned to one fixed counter for its lifetime (see {@link #threadCounterIndex}), + * so under high contention, threads split into up to 8 groups each contending on their own lock + * instead of all threads sharing one. This is the topology where {@code longAdderGroup}'s + * distributed locking should actually pay off, and where {@link Accumulator}'s thread-striped + * design has to earn its win on the write side rather than facing a single-counter worst case. + * {@code accumulateAndReset}/{@code groupAccumulateAnd} also now walk 8 slots per drain instead of + * 1, sizing the drain cost closer to {@code TracerHealthMetric}'s 54-constant production shape. + */ + enum Counter8 { + COUNTER_0, + COUNTER_1, + COUNTER_2, + COUNTER_3, + COUNTER_4, + COUNTER_5, + COUNTER_6, + COUNTER_7 + } + + private static final Counter8[] COUNTER8_VALUES = Counter8.values(); + private final LongAdder adder = new LongAdder(); private final Accumulator accumulator = Accumulator.of(Counter.values()); + private final Accumulator accumulator8 = Accumulator.of(Counter8.values()); private final ConcurrentHashMap chm = new ConcurrentHashMap<>(); private final LongAdder[] longAdderGroup = {new LongAdder()}; + private final LongAdder[] longAdderGroup8 = { + new LongAdder(), + new LongAdder(), + new LongAdder(), + new LongAdder(), + new LongAdder(), + new LongAdder(), + new LongAdder(), + new LongAdder() + }; + + /** + * Assigns each JMH worker thread a fixed {@code Counter8} index (round-robin over 8) the first + * time it calls into any {@code *8_*} benchmark, and keeps returning that same index for the + * thread's lifetime -- so under {@code Threads.MAX}, writes spread across all 8 counters instead + * of every thread hammering one. + */ + private final AtomicInteger threadIndexAssigner = new AtomicInteger(); + + private final ThreadLocal threadCounterIndex = + ThreadLocal.withInitial(() -> threadIndexAssigner.getAndIncrement() % COUNTER8_VALUES.length); /** * The natural "just use LongAdder" fix for the reset hazard: one {@code LongAdder} per counter, @@ -227,4 +288,56 @@ public void longAdderGroupAccumulateAnd_highContention(Blackhole blackhole) { groupInc(longAdderGroup, Counter.HITS.ordinal()); blackhole.consume(groupAccumulateAnd(longAdderGroup)); } + + @Benchmark + @Threads(1) + public void accumulatorIncrement8_lowContention() { + accumulator8.inc(COUNTER8_VALUES[threadCounterIndex.get()]); + } + + @Benchmark + @Threads(Threads.MAX) + public void accumulatorIncrement8_highContention() { + accumulator8.inc(COUNTER8_VALUES[threadCounterIndex.get()]); + } + + @Benchmark + @Threads(1) + public void accumulatorAccumulateAndReset8_lowContention(Blackhole blackhole) { + accumulator8.inc(COUNTER8_VALUES[threadCounterIndex.get()]); + blackhole.consume(accumulator8.accumulateAndReset()); + } + + @Benchmark + @Threads(Threads.MAX) + public void accumulatorAccumulateAndReset8_highContention(Blackhole blackhole) { + accumulator8.inc(COUNTER8_VALUES[threadCounterIndex.get()]); + blackhole.consume(accumulator8.accumulateAndReset()); + } + + @Benchmark + @Threads(1) + public void longAdderGroupIncrement8_lowContention() { + groupInc(longAdderGroup8, threadCounterIndex.get()); + } + + @Benchmark + @Threads(Threads.MAX) + public void longAdderGroupIncrement8_highContention() { + groupInc(longAdderGroup8, threadCounterIndex.get()); + } + + @Benchmark + @Threads(1) + public void longAdderGroupAccumulateAnd8_lowContention(Blackhole blackhole) { + groupInc(longAdderGroup8, threadCounterIndex.get()); + blackhole.consume(groupAccumulateAnd(longAdderGroup8)); + } + + @Benchmark + @Threads(Threads.MAX) + public void longAdderGroupAccumulateAnd8_highContention(Blackhole blackhole) { + groupInc(longAdderGroup8, threadCounterIndex.get()); + blackhole.consume(groupAccumulateAnd(longAdderGroup8)); + } } From b90e7f6718ad14cfe95b6b0aac9d27460c6b214f Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Thu, 10 Sep 2026 09:15:31 -0400 Subject: [PATCH 23/36] Remove Accumulator.of(E[])/Counts.zero(E[]) in favor of the Class factory The array overload let a caller pass a partial or reordered enum array, indexing writes by ordinal() while draining only values.length slots -- a counter could silently land in a slot no drain ever visits. Class always reads the canonical, complete enum() array, closing that gap. --- .../trace/util/AccumulatorBenchmark.java | 29 +++++++++-------- .../java/datadog/trace/util/Accumulator.java | 27 ++++------------ .../trace/util/AccumulatorFootprintTest.java | 4 +-- .../datadog/trace/util/AccumulatorTest.java | 32 +++++++------------ 4 files changed, 35 insertions(+), 57 deletions(-) diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java index 5345f490096..33b0bb80474 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java @@ -44,8 +44,8 @@ * lock to one lock shared by every thread -- the degenerate worst case for {@code longAdderGroup}, * with no thread-based distribution at all. The {@code *8_*} benchmarks fix that: each JMH worker * thread is pinned to one of 8 counters for its lifetime (see {@link #threadCounterIndex}), so - * {@code longAdderGroup8}'s threads split into up to 8 groups each contending their own lock -- - * the topology where distributed locking should actually pay off, forcing {@link Accumulator}'s + * {@code longAdderGroup8}'s threads split into up to 8 groups each contending their own lock -- the + * topology where distributed locking should actually pay off, forcing {@link Accumulator}'s * thread-striped design to earn its write-side win rather than facing a single-counter worst case. * Fork(5), 15 samples per benchmark: * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 15 2.746 ± 0.050 us/op @@ -66,9 +66,9 @@ * AccumulatorBenchmark.longAdderGroupIncrement8_lowContention avgt 15 0.020 ± 0.001 us/op * On the write side, {@link Accumulator} beats {@code longAdderGroup} at high contention by * ~255x in the degenerate single-shared-lock case and still by ~46x once counters are fairly spread - * across 8 locks -- a large, reproducible win either way, on the call that runs on every event. - * On the drain side, the two designs are close and the comparison is noisy under contention for - * both: at width 1 {@link Accumulator}'s drain (2.746 us/op) is actually faster than {@code + * across 8 locks -- a large, reproducible win either way, on the call that runs on every event. On + * the drain side, the two designs are close and the comparison is noisy under contention for both: + * at width 1 {@link Accumulator}'s drain (2.746 us/op) is actually faster than {@code * longAdderGroup}'s (4.770 ± 1.795 us/op, itself high-variance), and at width 8 it's only ~1.14x * slower (6.875 vs 6.025 us/op) -- not the regression an earlier reading of this benchmark * suggested. That earlier reading (13.357 us/op at Fork(2)) turned out to be a correlated anomaly @@ -93,13 +93,14 @@ enum Counter { * Unlike {@link Counter}, where every thread hits the single {@code HITS} constant (the worst * case for {@code longAdderGroup}'s per-counter locking -- one lock shared by every thread, * regardless of core count), these benchmarks spread writes across all 8 constants: each JMH - * worker thread is pinned to one fixed counter for its lifetime (see {@link #threadCounterIndex}), - * so under high contention, threads split into up to 8 groups each contending on their own lock - * instead of all threads sharing one. This is the topology where {@code longAdderGroup}'s - * distributed locking should actually pay off, and where {@link Accumulator}'s thread-striped - * design has to earn its win on the write side rather than facing a single-counter worst case. - * {@code accumulateAndReset}/{@code groupAccumulateAnd} also now walk 8 slots per drain instead of - * 1, sizing the drain cost closer to {@code TracerHealthMetric}'s 54-constant production shape. + * worker thread is pinned to one fixed counter for its lifetime (see {@link + * #threadCounterIndex}), so under high contention, threads split into up to 8 groups each + * contending on their own lock instead of all threads sharing one. This is the topology where + * {@code longAdderGroup}'s distributed locking should actually pay off, and where {@link + * Accumulator}'s thread-striped design has to earn its win on the write side rather than facing a + * single-counter worst case. {@code accumulateAndReset}/{@code groupAccumulateAnd} also now walk + * 8 slots per drain instead of 1, sizing the drain cost closer to {@code TracerHealthMetric}'s + * 54-constant production shape. */ enum Counter8 { COUNTER_0, @@ -115,8 +116,8 @@ enum Counter8 { private static final Counter8[] COUNTER8_VALUES = Counter8.values(); private final LongAdder adder = new LongAdder(); - private final Accumulator accumulator = Accumulator.of(Counter.values()); - private final Accumulator accumulator8 = Accumulator.of(Counter8.values()); + private final Accumulator accumulator = Accumulator.of(Counter.class); + private final Accumulator accumulator8 = Accumulator.of(Counter8.class); private final ConcurrentHashMap chm = new ConcurrentHashMap<>(); private final LongAdder[] longAdderGroup = {new LongAdder()}; private final LongAdder[] longAdderGroup8 = { diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index bb830b4acce..7bc246e4aed 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -12,7 +12,7 @@ *

{@code
  * enum MyCounters { FOO, BAR }
  *
- * Accumulator counters = Accumulator.of(MyCounters.values());
+ * Accumulator counters = Accumulator.of(MyCounters.class);
  * counters.inc(MyCounters.FOO);
  * counters.add(MyCounters.BAR, 5L);
  *
@@ -46,9 +46,10 @@ private Accumulator(AtomicLongArray[] data, int width, E[] values) {
   }
 
   /**
-   * @param values the enum constants naming each counter, e.g. {@code MyCounters.values()}
+   * @param enumType the enum naming each counter, e.g. {@code MyCounters.class}
    */
-  public static > Accumulator of(E[] values) {
+  public static > Accumulator of(Class enumType) {
+    E[] values = enumType.getEnumConstants();
     int width = values.length;
     int paddedWidth = paddedWidth(width);
     int stripes = stripeCount();
@@ -59,13 +60,6 @@ public static > Accumulator of(E[] values) {
     return new Accumulator<>(data, width, values);
   }
 
-  /**
-   * @param enumType the enum naming each counter, e.g. {@code MyCounters.class}
-   */
-  public static > Accumulator of(Class enumType) {
-    return of(enumType.getEnumConstants());
-  }
-
   /** Increments the counter named by {@code key} in the calling thread's stripe by one. */
   public void inc(E key) {
     add(key, 1L);
@@ -125,22 +119,15 @@ private Counts(long[] counts, E[] values) {
     }
 
     /**
-     * An all-zero {@link Counts}, sized for {@code values} -- for seeding a running total before
+     * An all-zero {@link Counts}, sized for {@code enumType} -- for seeding a running total before
      * any real drain has happened, without needing a scratch {@link Accumulator} just to call
      * {@link Accumulator#sum()} on it.
      *
-     * @param values the enum constants naming each counter, e.g. {@code MyCounters.values()}
-     */
-    public static > Counts zero(E[] values) {
-      return new Counts<>(new long[values.length], values);
-    }
-
-    /**
      * @param enumType the enum naming each counter, e.g. {@code MyCounters.class}
-     * @see #zero(Enum[])
      */
     public static > Counts zero(Class enumType) {
-      return zero(enumType.getEnumConstants());
+      E[] values = enumType.getEnumConstants();
+      return new Counts<>(new long[values.length], values);
     }
 
     /** The counter named by {@code key}. */
diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java
index a8f2f6f39d8..86b59be7960 100644
--- a/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java
+++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java
@@ -68,7 +68,7 @@ static LongAdder[] freshAdders() {
   @Test
   void freshFootprint() {
     LongAdder[] adders = freshAdders();
-    Accumulator accumulator = Accumulator.of(Counters.values());
+    Accumulator accumulator = Accumulator.of(Counters.class);
 
     long adderBytes = bytes((Object) adders);
     long accumulatorBytes = bytes(accumulator);
@@ -123,7 +123,7 @@ void contendedFootprint() throws InterruptedException {
     }
 
     long contendedAdderBytes = bytes((Object) adders);
-    Accumulator accumulator = Accumulator.of(Counters.values());
+    Accumulator accumulator = Accumulator.of(Counters.class);
     long accumulatorBytes = bytes(accumulator);
 
     System.out.printf(
diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java
index f17eab823fb..da433f3044d 100644
--- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java
+++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java
@@ -23,7 +23,7 @@ enum Counters {
 
   @Test
   void freshAccumulatorSumsToZero() {
-    Accumulator counters = Accumulator.of(Counters.values());
+    Accumulator counters = Accumulator.of(Counters.class);
     Accumulator.Counts drained = counters.accumulateAndReset();
     for (Counters c : Counters.values()) {
       assertEquals(0L, drained.get(c));
@@ -32,7 +32,7 @@ void freshAccumulatorSumsToZero() {
 
   @Test
   void incIncrementsByOne() {
-    Accumulator counters = Accumulator.of(Counters.values());
+    Accumulator counters = Accumulator.of(Counters.class);
     counters.inc(Counters.FOO);
     counters.inc(Counters.FOO);
     counters.inc(Counters.BAR);
@@ -45,7 +45,7 @@ void incIncrementsByOne() {
 
   @Test
   void addAppliesArbitraryDelta() {
-    Accumulator counters = Accumulator.of(Counters.values());
+    Accumulator counters = Accumulator.of(Counters.class);
     counters.add(Counters.BAZ, 41L);
     counters.add(Counters.BAZ, 1L);
 
@@ -55,7 +55,7 @@ void addAppliesArbitraryDelta() {
 
   @Test
   void accumulateAndResetsSoASecondDrainIsZero() {
-    Accumulator counters = Accumulator.of(Counters.values());
+    Accumulator counters = Accumulator.of(Counters.class);
     counters.inc(Counters.FOO);
 
     Accumulator.Counts first = counters.accumulateAndReset();
@@ -69,7 +69,7 @@ void accumulateAndResetsSoASecondDrainIsZero() {
 
   @Test
   void concurrentIncrementsAreNotLost() throws InterruptedException {
-    Accumulator counters = Accumulator.of(Counters.values());
+    Accumulator counters = Accumulator.of(Counters.class);
     int threadCount = 16;
     int incrementsPerThread = 10_000;
 
@@ -105,7 +105,7 @@ void concurrentIncrementsAreNotLost() throws InterruptedException {
   @Test
   void concurrentAccumulateAndDuringWritesNeverExceedsWritten()
       throws InterruptedException, ExecutionException, TimeoutException {
-    Accumulator counters = Accumulator.of(Counters.values());
+    Accumulator counters = Accumulator.of(Counters.class);
     int threadCount = 8;
     int incrementsPerThread = 5_000;
 
@@ -153,7 +153,7 @@ void concurrentAccumulateAndDuringWritesNeverExceedsWritten()
 
   @Test
   void sumDoesNotResetStripes() {
-    Accumulator counters = Accumulator.of(Counters.values());
+    Accumulator counters = Accumulator.of(Counters.class);
     counters.inc(Counters.FOO);
 
     Accumulator.Counts first = counters.sum();
@@ -170,7 +170,7 @@ void sumDoesNotResetStripes() {
 
   @Test
   void sumReflectsIncrementsMadeAfterAnEarlierSum() {
-    Accumulator counters = Accumulator.of(Counters.values());
+    Accumulator counters = Accumulator.of(Counters.class);
     counters.inc(Counters.FOO);
     counters.sum();
 
@@ -181,30 +181,20 @@ void sumReflectsIncrementsMadeAfterAnEarlierSum() {
 
   @Test
   void zeroSeedsAnAllZeroCountsWithoutAScratchAccumulator() {
-    Accumulator.Counts zero = Accumulator.Counts.zero(Counters.values());
+    Accumulator.Counts zero = Accumulator.Counts.zero(Counters.class);
     assertEquals(0L, zero.get(Counters.FOO));
     assertEquals(0L, zero.get(Counters.BAR));
 
-    Accumulator counters = Accumulator.of(Counters.values());
-    counters.inc(Counters.FOO);
-
-    Accumulator.Counts live = zero.plus(counters.sum());
-    assertEquals(1L, live.get(Counters.FOO));
-  }
-
-  @Test
-  void ofAndZeroAcceptAnEnumClassInsteadOfAValuesArray() {
     Accumulator counters = Accumulator.of(Counters.class);
     counters.inc(Counters.FOO);
 
-    Accumulator.Counts zero = Accumulator.Counts.zero(Counters.class);
     Accumulator.Counts live = zero.plus(counters.sum());
     assertEquals(1L, live.get(Counters.FOO));
   }
 
   @Test
   void countsExposesItsOwnKeysWithoutASeparateValuesArray() {
-    Accumulator counters = Accumulator.of(Counters.values());
+    Accumulator counters = Accumulator.of(Counters.class);
     counters.inc(Counters.FOO);
     counters.add(Counters.BAR, 5L);
 
@@ -220,7 +210,7 @@ void countsExposesItsOwnKeysWithoutASeparateValuesArray() {
 
   @Test
   void plusCombinesAStoredRunningTotalWithALiveSumWithoutMutatingEither() {
-    Accumulator counters = Accumulator.of(Counters.values());
+    Accumulator counters = Accumulator.of(Counters.class);
     counters.inc(Counters.FOO);
     counters.add(Counters.BAR, 5L);
 

From 97274e55ce4ba463900b0e9234b4753a3f8f1940 Mon Sep 17 00:00:00 2001
From: Douglas Q Hawkins 
Date: Thu, 10 Sep 2026 09:57:14 -0400
Subject: [PATCH 24/36] Add unstriped AtomicLongArray baseline to
 AccumulatorBenchmark

Isolates the cost of Accumulator's thread-striping from the cost of
correctness: one shared AtomicLongArray, getAndAdd for writes and
getAndSet(0) for drain, no locking and no per-thread distribution at
all. Numbers not yet re-measured cleanly -- a quick sanity run confirms
the new benchmarks execute, but a proper Fork(5) reading is pending.
---
 .../trace/util/AccumulatorBenchmark.java      | 89 +++++++++++++++++--
 1 file changed, 81 insertions(+), 8 deletions(-)

diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java
index 33b0bb80474..dd3467a668e 100644
--- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java
+++ b/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java
@@ -5,6 +5,7 @@
 import java.util.concurrent.ConcurrentHashMap;
 import java.util.concurrent.atomic.AtomicInteger;
 import java.util.concurrent.atomic.AtomicLong;
+import java.util.concurrent.atomic.AtomicLongArray;
 import java.util.concurrent.atomic.LongAdder;
 import org.openjdk.jmh.annotations.Benchmark;
 import org.openjdk.jmh.annotations.BenchmarkMode;
@@ -26,14 +27,17 @@
  * counter guarded by a per-counter lock (the "just fix it with LongAdder" natural migration target
  * -- {@code longAdderGroup*}), and the {@code ConcurrentHashMap.computeIfAbsent(key, k -> new
  * AtomicLong())} anti-pattern ({@code chmAtomicLongIncrement*}) that {@link Accumulator} exists to
- * avoid. The CHM variant allocates its counter under the bucket's bin lock the first time its one
- * constant key is seen, but since the map is a {@code @State(Scope.Benchmark)} field shared across
- * the whole run, that allocation happens exactly once; every sampled op after it hits the warmed,
- * already-present fast path. So this measures steady-state {@code computeIfAbsent} lookup overhead
- * on an already-populated map, not the one-time allocation-under-lock cost -- still a useful number
- * (a fixed, small key set that's allocated once and hit for the life of the process, as {@code
- * WafMetricCollector}-style CHM counters are, spends nearly all its time in this same warmed path),
- * just not the pathology the name of this benchmark might suggest.
+ * avoid, and an unstriped {@code AtomicLongArray} ({@code atomicLongArray*}) -- one shared array,
+ * no per-thread distribution, isolating the cost of striping itself from the cost of correctness
+ * (see {@link #arrayAccumulateAndReset}). The CHM variant allocates its counter under the bucket's
+ * bin lock the first time its one constant key is seen, but since the map is a
+ * {@code @State(Scope.Benchmark)} field shared across the whole run, that allocation happens
+ * exactly once; every sampled op after it hits the warmed, already-present fast path. So this
+ * measures steady-state {@code computeIfAbsent} lookup overhead on an already-populated map, not
+ * the one-time allocation-under-lock cost -- still a useful number (a fixed, small key set that's
+ * allocated once and hit for the life of the process, as {@code WafMetricCollector}-style CHM
+ * counters are, spends nearly all its time in this same warmed path), just not the pathology the
+ * name of this benchmark might suggest.
  *
  * 

{@code longAdderGroup*}: is a "just fix it with LongAdder" helper actually cheaper? * {@code groupInc}/{@code groupAccumulateAnd} are the natural correct fix using {@code LongAdder} @@ -119,6 +123,8 @@ enum Counter8 { private final Accumulator accumulator = Accumulator.of(Counter.class); private final Accumulator accumulator8 = Accumulator.of(Counter8.class); private final ConcurrentHashMap chm = new ConcurrentHashMap<>(); + private final AtomicLongArray atomicLongArray = new AtomicLongArray(1); + private final AtomicLongArray atomicLongArray8 = new AtomicLongArray(8); private final LongAdder[] longAdderGroup = {new LongAdder()}; private final LongAdder[] longAdderGroup8 = { new LongAdder(), @@ -170,6 +176,21 @@ private static long[] groupAccumulateAnd(LongAdder[] group) { return acc; } + /** + * The unstriped baseline: a single shared {@code AtomicLongArray}, one slot per counter, with no + * per-thread distribution at all -- isolates the cost of {@link Accumulator}'s thread-striping + * itself from the cost of correctness (unlike {@code longAdderGroup}, this has no lock: {@code + * getAndAdd} and {@code getAndSet} are each already atomic per-slot, so no coordination is needed + * to give the same "no increment lost across a drain" guarantee). + */ + private static long[] arrayAccumulateAndReset(AtomicLongArray array) { + long[] acc = new long[array.length()]; + for (int i = 0; i < array.length(); i++) { + acc[i] = array.getAndSet(i, 0L); + } + return acc; + } + @Benchmark @Threads(1) public void longAdderIncrement_lowContention() { @@ -290,6 +311,58 @@ public void longAdderGroupAccumulateAnd_highContention(Blackhole blackhole) { blackhole.consume(groupAccumulateAnd(longAdderGroup)); } + @Benchmark + @Threads(1) + public void atomicLongArrayIncrement_lowContention() { + atomicLongArray.getAndAdd(Counter.HITS.ordinal(), 1L); + } + + @Benchmark + @Threads(Threads.MAX) + public void atomicLongArrayIncrement_highContention() { + atomicLongArray.getAndAdd(Counter.HITS.ordinal(), 1L); + } + + @Benchmark + @Threads(1) + public void atomicLongArrayAccumulateAndReset_lowContention(Blackhole blackhole) { + atomicLongArray.getAndAdd(Counter.HITS.ordinal(), 1L); + blackhole.consume(arrayAccumulateAndReset(atomicLongArray)); + } + + @Benchmark + @Threads(Threads.MAX) + public void atomicLongArrayAccumulateAndReset_highContention(Blackhole blackhole) { + atomicLongArray.getAndAdd(Counter.HITS.ordinal(), 1L); + blackhole.consume(arrayAccumulateAndReset(atomicLongArray)); + } + + @Benchmark + @Threads(1) + public void atomicLongArrayIncrement8_lowContention() { + atomicLongArray8.getAndAdd(threadCounterIndex.get(), 1L); + } + + @Benchmark + @Threads(Threads.MAX) + public void atomicLongArrayIncrement8_highContention() { + atomicLongArray8.getAndAdd(threadCounterIndex.get(), 1L); + } + + @Benchmark + @Threads(1) + public void atomicLongArrayAccumulateAndReset8_lowContention(Blackhole blackhole) { + atomicLongArray8.getAndAdd(threadCounterIndex.get(), 1L); + blackhole.consume(arrayAccumulateAndReset(atomicLongArray8)); + } + + @Benchmark + @Threads(Threads.MAX) + public void atomicLongArrayAccumulateAndReset8_highContention(Blackhole blackhole) { + atomicLongArray8.getAndAdd(threadCounterIndex.get(), 1L); + blackhole.consume(arrayAccumulateAndReset(atomicLongArray8)); + } + @Benchmark @Threads(1) public void accumulatorIncrement8_lowContention() { From f28a67adf9895778a0dc9a3d39cc9bfba7e28a32 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Thu, 10 Sep 2026 10:25:46 -0400 Subject: [PATCH 25/36] Add Accumulator.RunningTotal, an opt-in drain+live-read composition TracerHealthMetrics needs both a periodic destructive drain (for statsd reporting) and a separate live diagnostic read (summary()), coordinated so the two never observe a torn state across a drain. RunningTotal encapsulates that one-lock pattern as a reusable, opt-in layer on top of Accumulator, which stays the primary primitive for the more common drain-only/live-only cases. Co-Authored-By: Claude Sonnet 5 --- .../java/datadog/trace/util/Accumulator.java | 59 +++++++++++++++++++ .../datadog/trace/util/AccumulatorTest.java | 39 ++++++++++++ 2 files changed, 98 insertions(+) diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index 7bc246e4aed..2fa01614e0a 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -158,6 +158,65 @@ public Counts plus(Counts other) { } } + /** + * An opt-in, safe-by-default composition for the one recurring shape {@link Accumulator} itself + * deliberately doesn't try to make safe on its own: a periodic destructive drain for reporting + * (e.g. to statsd) alongside a separate, non-destructive live read for diagnostics (e.g. a {@code + * summary()} or {@code toString()}). Read independently, {@link #drain} and {@link #live} are + * each individually correct -- the hazard is between them: a {@link #live} call landing between a + * {@link #drain}'s reset and its caller publishing the delta into a running total would combine + * the pre-publish total with the post-drain (zeroed) {@link #sum}, silently under-reporting by + * the just-drained delta. {@code RunningTotal} closes that gap with one lock shared by {@link + * #drain} and {@link #live} -- most callers don't need this (a plain drain-only reporter, or a + * live-read-only diagnostic, needs no coordination at all), so it's a separate, opt-in type + * rather than baked into every {@link Accumulator}. + */ + public static final class RunningTotal> { + private final Accumulator accumulator; + private final Object lock = new Object(); + private Counts total; + + private RunningTotal(Accumulator accumulator) { + this.accumulator = accumulator; + // accumulateAndReset(), not sum() -- sum() doesn't reset, so seeding from it would double + // count whatever's already there once a later live() adds a fresh sum() on top of it. + this.total = accumulator.accumulateAndReset(); + } + + /** + * Wraps {@code accumulator}, seeding the running total from a drain of whatever it currently + * holds (typically nothing, for a freshly constructed {@code accumulator}). + */ + public static > RunningTotal of(Accumulator accumulator) { + return new RunningTotal<>(accumulator); + } + + /** + * Atomically drains the accumulator and folds the delta into the running total -- call this + * right before reporting the delta to a downstream sink on a reporting cadence. + * + * @return the delta just drained + */ + public Counts drain() { + synchronized (lock) { + Counts delta = accumulator.accumulateAndReset(); + total = total.plus(delta); + return delta; + } + } + + /** + * The live total: the running total as of the last {@link #drain}, folded with whatever's + * accumulated since -- never resets anything, safe to call at any time without perturbing a + * concurrent {@link #drain}. + */ + public Counts live() { + synchronized (lock) { + return total.plus(accumulator.sum()); + } + } + } + /** * The calling thread's stripe: cheap masking, no allocation, no map lookup. * diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java index da433f3044d..40eb3237409 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java @@ -228,4 +228,43 @@ void plusCombinesAStoredRunningTotalWithALiveSumWithoutMutatingEither() { assertEquals(1L, storedTotal.get(Counters.FOO)); assertEquals(1L, counters.sum().get(Counters.FOO)); } + + @Test + void runningTotalSeedsFromTheAccumulatorsCurrentSum() { + Accumulator counters = Accumulator.of(Counters.class); + counters.inc(Counters.FOO); + + Accumulator.RunningTotal runningTotal = Accumulator.RunningTotal.of(counters); + assertEquals(1L, runningTotal.live().get(Counters.FOO)); + } + + @Test + void runningTotalDrainReturnsTheDeltaAndFoldsItIntoTheTotal() { + Accumulator counters = Accumulator.of(Counters.class); + Accumulator.RunningTotal runningTotal = Accumulator.RunningTotal.of(counters); + + counters.inc(Counters.FOO); + Accumulator.Counts delta = runningTotal.drain(); + assertEquals(1L, delta.get(Counters.FOO)); + assertEquals(1L, runningTotal.live().get(Counters.FOO)); + + counters.inc(Counters.FOO); + Accumulator.Counts secondDelta = runningTotal.drain(); + assertEquals(1L, secondDelta.get(Counters.FOO)); + assertEquals(2L, runningTotal.live().get(Counters.FOO)); + } + + @Test + void runningTotalLiveReflectsActivitySinceTheLastDrainWithoutDraining() { + Accumulator counters = Accumulator.of(Counters.class); + Accumulator.RunningTotal runningTotal = Accumulator.RunningTotal.of(counters); + + counters.inc(Counters.FOO); + runningTotal.drain(); + + counters.inc(Counters.FOO); + assertEquals(2L, runningTotal.live().get(Counters.FOO)); + // live() didn't drain anything, so a real drain afterwards still sees the pending increment + assertEquals(1L, runningTotal.drain().get(Counters.FOO)); + } } From 631eddeb0af0a054c249d711f31dcfe7c3b21306 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Thu, 10 Sep 2026 10:40:46 -0400 Subject: [PATCH 26/36] Rename Accumulator/Counts field values -> keys The keys() accessor already used that name; the backing field lagged behind. Co-Authored-By: Claude Sonnet 5 --- .../java/datadog/trace/util/Accumulator.java | 30 +++++++++---------- 1 file changed, 15 insertions(+), 15 deletions(-) diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/internal-api/src/main/java/datadog/trace/util/Accumulator.java index 2fa01614e0a..2216531d09d 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/internal-api/src/main/java/datadog/trace/util/Accumulator.java @@ -37,27 +37,27 @@ public final class Accumulator> { private final AtomicLongArray[] data; private final int width; - private final E[] values; + private final E[] keys; - private Accumulator(AtomicLongArray[] data, int width, E[] values) { + private Accumulator(AtomicLongArray[] data, int width, E[] keys) { this.data = data; this.width = width; - this.values = values; + this.keys = keys; } /** * @param enumType the enum naming each counter, e.g. {@code MyCounters.class} */ public static > Accumulator of(Class enumType) { - E[] values = enumType.getEnumConstants(); - int width = values.length; + E[] keys = enumType.getEnumConstants(); + int width = keys.length; int paddedWidth = paddedWidth(width); int stripes = stripeCount(); AtomicLongArray[] data = new AtomicLongArray[stripes]; for (int i = 0; i < stripes; i++) { data[i] = new AtomicLongArray(paddedWidth); } - return new Accumulator<>(data, width, values); + return new Accumulator<>(data, width, keys); } /** Increments the counter named by {@code key} in the calling thread's stripe by one. */ @@ -84,7 +84,7 @@ public Counts accumulateAndReset() { acc[i] += stripe.getAndSet(i, 0L); } } - return new Counts<>(acc, values); + return new Counts<>(acc, keys); } /** @@ -101,7 +101,7 @@ public Counts sum() { acc[i] += stripe.get(i); } } - return new Counts<>(acc, values); + return new Counts<>(acc, keys); } /** @@ -111,11 +111,11 @@ public Counts sum() { */ public static final class Counts> { private final long[] counts; - private final E[] values; + private final E[] keys; - private Counts(long[] counts, E[] values) { + private Counts(long[] counts, E[] keys) { this.counts = counts; - this.values = values; + this.keys = keys; } /** @@ -126,8 +126,8 @@ private Counts(long[] counts, E[] values) { * @param enumType the enum naming each counter, e.g. {@code MyCounters.class} */ public static > Counts zero(Class enumType) { - E[] values = enumType.getEnumConstants(); - return new Counts<>(new long[values.length], values); + E[] keys = enumType.getEnumConstants(); + return new Counts<>(new long[keys.length], keys); } /** The counter named by {@code key}. */ @@ -141,7 +141,7 @@ public long get(E key) { * {@code E.values()} alongside this object. */ public E[] keys() { - return values; + return keys; } /** @@ -154,7 +154,7 @@ public Counts plus(Counts other) { for (int i = 0; i < combined.length; i++) { combined[i] += other.counts[i]; } - return new Counts<>(combined, values); + return new Counts<>(combined, keys); } } From 8e6e624d5d2dc9d38a978d41669a86f87fb85580 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Thu, 10 Sep 2026 11:30:40 -0400 Subject: [PATCH 27/36] Move Accumulator from internal-api into metrics-api, next to Counter Accumulator and Counter are the same conceptual primitive at different cost points: Counter's real implementation (StatsDCounter) forwards synchronously to statsd on every increment, while Accumulator stripes by thread and defers reporting to a periodic drain. Colocating them in metrics-api makes that choice visible instead of crossing an internal-api/metrics-api boundary via the Accumulator.Counts overload on StatsDCountReporter. Also adds AccumulatorVsCounterBenchmark, a decision-support benchmark comparing Accumulator against a Counter implementation mirroring StatsDCounter's real synchronous-statsd-call shape (backed by StatsDClient.NO_OP to isolate from network I/O). This is distinct from the existing AccumulatorBenchmark, which is a design-proof benchmark justifying Accumulator's internal striping strategy against synthetic alternatives (LongAdder, CHM, unstriped AtomicLongArray) -- a question already settled during Accumulator's initial development. Co-Authored-By: Claude Sonnet 5 --- products/metrics/metrics-api/build.gradle.kts | 10 ++ .../metrics/api}/AccumulatorBenchmark.java | 2 +- .../api/AccumulatorVsCounterBenchmark.java | 107 ++++++++++++++++++ .../datadog/metrics/api}/Accumulator.java | 2 +- .../api}/AccumulatorFootprintTest.java | 2 +- .../datadog/metrics/api}/AccumulatorTest.java | 2 +- 6 files changed, 121 insertions(+), 4 deletions(-) rename {internal-api/src/jmh/java/datadog/trace/util => products/metrics/metrics-api/src/jmh/java/datadog/metrics/api}/AccumulatorBenchmark.java (99%) create mode 100644 products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorVsCounterBenchmark.java rename {internal-api/src/main/java/datadog/trace/util => products/metrics/metrics-api/src/main/java/datadog/metrics/api}/Accumulator.java (99%) rename {internal-api/src/test/java/datadog/trace/util => products/metrics/metrics-api/src/test/java/datadog/metrics/api}/AccumulatorFootprintTest.java (99%) rename {internal-api/src/test/java/datadog/trace/util => products/metrics/metrics-api/src/test/java/datadog/metrics/api}/AccumulatorTest.java (99%) diff --git a/products/metrics/metrics-api/build.gradle.kts b/products/metrics/metrics-api/build.gradle.kts index bc995a5c87d..dec7a091151 100644 --- a/products/metrics/metrics-api/build.gradle.kts +++ b/products/metrics/metrics-api/build.gradle.kts @@ -1,10 +1,20 @@ plugins { `java-library` id("dd-trace-java.module.internal-api") + id("dd-trace-java.jmh-conventions") } description = "Metrics API" dependencies { implementation(libs.slf4j) + api(project(":components:environment")) + + testImplementation(libs.bundles.junit5) + testImplementation(libs.jol.core) +} + +jmh { + jmhVersion = libs.versions.jmh.get() + duplicateClassesStrategy = DuplicatesStrategy.EXCLUDE } diff --git a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java b/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorBenchmark.java similarity index 99% rename from internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java rename to products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorBenchmark.java index dd3467a668e..d7c1779165f 100644 --- a/internal-api/src/jmh/java/datadog/trace/util/AccumulatorBenchmark.java +++ b/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorBenchmark.java @@ -1,4 +1,4 @@ -package datadog.trace.util; +package datadog.metrics.api; import static java.util.concurrent.TimeUnit.MICROSECONDS; diff --git a/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorVsCounterBenchmark.java b/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorVsCounterBenchmark.java new file mode 100644 index 00000000000..d0b600871d7 --- /dev/null +++ b/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorVsCounterBenchmark.java @@ -0,0 +1,107 @@ +package datadog.metrics.api; + +import static java.util.concurrent.TimeUnit.MICROSECONDS; + +import datadog.metrics.api.statsd.StatsDClient; +import org.openjdk.jmh.annotations.Benchmark; +import org.openjdk.jmh.annotations.BenchmarkMode; +import org.openjdk.jmh.annotations.Fork; +import org.openjdk.jmh.annotations.Measurement; +import org.openjdk.jmh.annotations.Mode; +import org.openjdk.jmh.annotations.OutputTimeUnit; +import org.openjdk.jmh.annotations.Scope; +import org.openjdk.jmh.annotations.State; +import org.openjdk.jmh.annotations.Threads; +import org.openjdk.jmh.annotations.Warmup; + +/** + * A decision-support benchmark, not a design-proof one: {@link AccumulatorBenchmark} exists to + * justify {@link Accumulator}'s internal striping design against synthetic alternatives ({@code + * LongAdder}, a CHM of {@code AtomicLong}, an unstriped {@code AtomicLongArray}) that nobody in + * this codebase would actually reach for instead -- that question is settled and doesn't need + * re-litigating on every read. This class answers a different, durable one: given both {@link + * Counter} and {@link Accumulator} live in {@code metrics-api}, which do you actually use? + * + *

{@link Counter} is the one most callers reach for first, and for good reason -- it's the + * advertised, general-purpose metrics API. But its real implementation ({@code StatsDCounter} in + * {@code metrics-lib}) calls {@link StatsDClient#count} synchronously on every {@link + * Counter#increment}, with no batching or striping of its own. {@link #counterIncrement} below + * mirrors that shape exactly (forwarding straight to a no-op {@link StatsDClient} to isolate the + * call-site cost from real network I/O, which would swamp everything else and isn't the question + * this asks -- {@code StatsDCounter} itself has package-private construction, so this reimplements + * its shape rather than depending on {@code metrics-lib}). {@link Accumulator} exists because that + * per-call cost is too high to pay on every request/span/event -- it stripes by thread and defers + * reporting to a periodic drain instead. + * + *

Rule of thumb: a counter incremented on a hot path (every request, span, or event) + * should use {@link Accumulator}, drained on a reporting cadence. A counter incremented rarely + * (startup, config changes, an error path already off the hot path) can use {@link Counter} + * directly, no ceremony required. This is the concrete case behind the (not yet built) + * {@code @ForegroundSafe}/{@code @BackgroundOnly} annotations: {@link Counter#increment} is + * {@code @BackgroundOnly}, {@link Accumulator#inc} is {@code @ForegroundSafe}. + */ +@State(Scope.Benchmark) +@Warmup(iterations = 1, time = 10) +@Measurement(iterations = 3, time = 10) +@BenchmarkMode(Mode.AverageTime) +@OutputTimeUnit(MICROSECONDS) +@Fork(5) +public class AccumulatorVsCounterBenchmark { + + enum Metric { + HITS + } + + private final Accumulator accumulator = Accumulator.of(Metric.class); + private final Counter counter = new SynchronousStatsDCounter("hits", StatsDClient.NO_OP); + + /** + * Mirrors {@code StatsDCounter}'s real shape: every {@link #increment} forwards straight to the + * client, with no batching of its own. Reimplemented here rather than depending on {@code + * metrics-lib} because {@code StatsDCounter}'s constructor is package-private. + */ + private static final class SynchronousStatsDCounter implements Counter { + private static final String[] NO_TAGS = new String[0]; + private final String name; + private final StatsDClient statsd; + + SynchronousStatsDCounter(String name, StatsDClient statsd) { + this.name = name; + this.statsd = statsd; + } + + @Override + public void increment(int delta) { + statsd.count(name, delta, NO_TAGS); + } + + @Override + public void incrementErrorCount(String cause, int delta) { + statsd.count(name, delta, new String[] {"cause:" + cause}); + } + } + + @Benchmark + @Threads(1) + public void accumulatorIncrement_lowContention() { + accumulator.inc(Metric.HITS); + } + + @Benchmark + @Threads(Threads.MAX) + public void accumulatorIncrement_highContention() { + accumulator.inc(Metric.HITS); + } + + @Benchmark + @Threads(1) + public void counterIncrement_lowContention() { + counter.increment(1); + } + + @Benchmark + @Threads(Threads.MAX) + public void counterIncrement_highContention() { + counter.increment(1); + } +} diff --git a/internal-api/src/main/java/datadog/trace/util/Accumulator.java b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java similarity index 99% rename from internal-api/src/main/java/datadog/trace/util/Accumulator.java rename to products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java index 2216531d09d..b991fb613d9 100644 --- a/internal-api/src/main/java/datadog/trace/util/Accumulator.java +++ b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java @@ -1,4 +1,4 @@ -package datadog.trace.util; +package datadog.metrics.api; import datadog.environment.ThreadSupport; import java.util.concurrent.atomic.AtomicLongArray; diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorFootprintTest.java similarity index 99% rename from internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java rename to products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorFootprintTest.java index 86b59be7960..b0b2f7b59fd 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorFootprintTest.java +++ b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorFootprintTest.java @@ -1,4 +1,4 @@ -package datadog.trace.util; +package datadog.metrics.api; import static org.junit.jupiter.api.Assertions.assertTrue; import static org.junit.jupiter.api.Assumptions.assumeFalse; diff --git a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java similarity index 99% rename from internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java rename to products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java index 40eb3237409..90b78895b21 100644 --- a/internal-api/src/test/java/datadog/trace/util/AccumulatorTest.java +++ b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java @@ -1,4 +1,4 @@ -package datadog.trace.util; +package datadog.metrics.api; import static org.junit.jupiter.api.Assertions.assertEquals; import static org.junit.jupiter.api.Assertions.assertTrue; From b3075dbb781baf48fd1d94a6f1eb939ce0ecbe97 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Thu, 10 Sep 2026 13:27:17 -0400 Subject: [PATCH 28/36] Mark Accumulator/RunningTotal @ThreadSafe; add Counts.from(fromIndex) Counts.from(fromIndex) zeroes every entry before the given index, so a caller that partially delivered a drained batch downstream can represent "what's left to retry" without re-counting the part that already made it. Used by StatsDCountReporter's compensation logic in the TracerHealthMetrics migration (dougqh/accumulator-tracerhealthmetrics). Co-Authored-By: Claude Sonnet 5 --- .../java/datadog/metrics/api/Accumulator.java | 14 ++++++++++++++ .../datadog/metrics/api/AccumulatorTest.java | 18 ++++++++++++++++++ 2 files changed, 32 insertions(+) diff --git a/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java index b991fb613d9..6a44390d405 100644 --- a/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java +++ b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java @@ -2,6 +2,7 @@ import datadog.environment.ThreadSupport; import java.util.concurrent.atomic.AtomicLongArray; +import javax.annotation.concurrent.ThreadSafe; /** * A striped, lock-free counter primitive keyed by enum ordinal: {@code LongAdder}'s write @@ -31,6 +32,7 @@ * unrelated aspects of the same event, not maintaining a cross-counter invariant a reader depends * on) or bring their own coordination. */ +@ThreadSafe public final class Accumulator> { /** One full cache line of {@code long}s (64 bytes), used to pad each stripe row. */ private static final int CACHE_LINE_LONGS = 8; @@ -156,6 +158,17 @@ public Counts plus(Counts other) { } return new Counts<>(combined, keys); } + + /** + * A copy of this {@link Counts} with every entry before {@code keys()[fromIndex]} zeroed out -- + * for a caller that partially delivered a batch downstream and wants to represent "what's left" + * to retry, without re-counting the part that already made it. + */ + public Counts from(int fromIndex) { + long[] remaining = new long[counts.length]; + System.arraycopy(counts, fromIndex, remaining, fromIndex, counts.length - fromIndex); + return new Counts<>(remaining, keys); + } } /** @@ -171,6 +184,7 @@ public Counts plus(Counts other) { * live-read-only diagnostic, needs no coordination at all), so it's a separate, opt-in type * rather than baked into every {@link Accumulator}. */ + @ThreadSafe public static final class RunningTotal> { private final Accumulator accumulator; private final Object lock = new Object(); diff --git a/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java index 90b78895b21..3d2592fd522 100644 --- a/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java +++ b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java @@ -229,6 +229,24 @@ void plusCombinesAStoredRunningTotalWithALiveSumWithoutMutatingEither() { assertEquals(1L, counters.sum().get(Counters.FOO)); } + @Test + void fromZeroesEveryEntryBeforeTheGivenIndex() { + Accumulator counters = Accumulator.of(Counters.class); + counters.inc(Counters.FOO); + counters.add(Counters.BAR, 4L); + counters.add(Counters.BAZ, 2L); + + Accumulator.Counts drained = counters.accumulateAndReset(); + Accumulator.Counts remaining = drained.from(Counters.BAR.ordinal()); + + assertEquals(0L, remaining.get(Counters.FOO)); + assertEquals(4L, remaining.get(Counters.BAR)); + assertEquals(2L, remaining.get(Counters.BAZ)); + + // the original Counts is untouched by taking a remainder from it + assertEquals(1L, drained.get(Counters.FOO)); + } + @Test void runningTotalSeedsFromTheAccumulatorsCurrentSum() { Accumulator counters = Accumulator.of(Counters.class); From 108ce6916927cc434ae7995f05ebac96f8e830c4 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Thu, 10 Sep 2026 13:40:03 -0400 Subject: [PATCH 29/36] Add StatsDCountReporter: an Accumulator-backed periodic statsd reporter with retry Owns an Accumulator + RunningTotal for one enum's counters, exposing inc/add/live/flush. flush() drains and reports the delta, retrying whatever a statsd exception left undelivered on the next cycle (Counts.from(i)) instead of dropping it -- without perturbing the cumulative live() total, since the drain already folded the delta into it unconditionally before delivery was attempted. Co-Authored-By: Claude Sonnet 5 --- .../api/statsd/StatsDCountReporter.java | 121 ++++++++++ .../metrics/api/statsd/StatsDCounterKey.java | 8 + .../api/statsd/RecordingStatsDClient.java | 63 ++++++ .../api/statsd/StatsDCountReporterTest.java | 207 ++++++++++++++++++ 4 files changed, 399 insertions(+) create mode 100644 products/metrics/metrics-api/src/main/java/datadog/metrics/api/statsd/StatsDCountReporter.java create mode 100644 products/metrics/metrics-api/src/main/java/datadog/metrics/api/statsd/StatsDCounterKey.java create mode 100644 products/metrics/metrics-api/src/test/java/datadog/metrics/api/statsd/RecordingStatsDClient.java create mode 100644 products/metrics/metrics-api/src/test/java/datadog/metrics/api/statsd/StatsDCountReporterTest.java diff --git a/products/metrics/metrics-api/src/main/java/datadog/metrics/api/statsd/StatsDCountReporter.java b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/statsd/StatsDCountReporter.java new file mode 100644 index 00000000000..0017039c80a --- /dev/null +++ b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/statsd/StatsDCountReporter.java @@ -0,0 +1,121 @@ +package datadog.metrics.api.statsd; + +import datadog.metrics.api.Accumulator; +import java.util.function.ToLongFunction; +import javax.annotation.concurrent.ThreadSafe; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +/** + * Owns an {@link Accumulator} plus the periodic destructive-drain-and-report cycle over it: {@link + * #inc}/{@link #add} write straight through to the accumulator, {@link #live} peeks it for a + * diagnostic read, and {@link #flush} drains it and reports the delta to a {@link StatsDClient} -- + * holding onto whatever wasn't delivered and retrying it on the next {@link #flush}, rather than + * losing it. + * + *

This isn't a real transaction: statsd delivery is already best-effort over UDP, so there's no + * rollback on the wire and no way to recover a packet actually lost in transit. It only protects + * against a local exception (a bad client, a malformed tag, a config error -- not a lost packet) + * turning "this flush interval under-reports" into "this delta is gone forever." + * + *

The undelivered remainder is kept as {@link #pending} state on this object, not fed back into + * the {@link Accumulator} -- {@link Accumulator.RunningTotal#drain} already folds every drained + * delta into its cumulative total unconditionally, before delivery is attempted, since that total + * counts real events, independent of whether statsd ever got them. Re-adding the same delta to the + * accumulator would make a later {@link Accumulator.RunningTotal#drain} see it as new and count it + * a second time -- {@link #pending} carries it forward for statsd's benefit only. {@link #flush} is + * assumed single-threaded (it's the body of one periodic scheduled task), so no locking guards it. + */ +@ThreadSafe +public final class StatsDCountReporter & StatsDCounterKey> { + private static final Logger log = LoggerFactory.getLogger(StatsDCountReporter.class); + + private final StatsDClient statsDClient; + private final Accumulator accumulator; + private final Accumulator.RunningTotal runningTotal; + + /** Undelivered from the last {@link #flush}, to retry on the next one; {@code null} if none. */ + private Accumulator.Counts pending; + + /** + * @param enumType the enum naming each counter, e.g. {@code MyCounters.class} + */ + public static & StatsDCounterKey> StatsDCountReporter of( + StatsDClient statsDClient, Class enumType) { + return new StatsDCountReporter<>(statsDClient, Accumulator.of(enumType)); + } + + private StatsDCountReporter(StatsDClient statsDClient, Accumulator accumulator) { + this.statsDClient = statsDClient; + this.accumulator = accumulator; + this.runningTotal = Accumulator.RunningTotal.of(accumulator); + } + + /** Increments the counter named by {@code key} by one. */ + public void inc(E key) { + accumulator.inc(key); + } + + /** Adds {@code delta} to the counter named by {@code key}. */ + public void add(E key, long delta) { + accumulator.add(key, delta); + } + + /** + * The live total -- see {@link Accumulator.RunningTotal#live}. For a diagnostic read (e.g. a + * {@code summary()}), never for deciding what to report: {@link #flush} owns that. + */ + public Accumulator.Counts live() { + return runningTotal.live(); + } + + /** + * Drains the accumulator and reports the delta (plus anything still owed from a prior failed + * attempt), holding onto whatever isn't confirmed delivered this time for the next {@link + * #flush}. + */ + public void flush() { + Accumulator.Counts drained = runningTotal.drain(); + Accumulator.Counts toReport = pending == null ? drained : pending.plus(drained); + pending = report(toReport); + } + + /** + * @return {@code null} if every counter was delivered, otherwise a {@link Accumulator.Counts} + * holding whatever wasn't attempted or confirmed sent + */ + private Accumulator.Counts report(Accumulator.Counts counts) { + E[] keys = counts.keys(); + for (int i = 0; i < keys.length; i++) { + E key = keys[i]; + long delta = counts.get(key); + if (delta != 0) { + try { + statsDClient.count(key.getMetricName(), delta, key.getTags()); + } catch (RuntimeException e) { + log.debug( + "Failed to report {}, compensating {} undelivered counter(s)", + key, + keys.length - i, + e); + return counts.from(i); + } + } + } + return null; + } + + /** + * Reports every nonzero entry of {@code values}/{@code counts} directly -- no accumulator, no + * compensation. + */ + public static & StatsDCounterKey> void report( + StatsDClient statsDClient, E[] values, ToLongFunction counts) { + for (E value : values) { + long delta = counts.applyAsLong(value); + if (delta != 0) { + statsDClient.count(value.getMetricName(), delta, value.getTags()); + } + } + } +} diff --git a/products/metrics/metrics-api/src/main/java/datadog/metrics/api/statsd/StatsDCounterKey.java b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/statsd/StatsDCounterKey.java new file mode 100644 index 00000000000..3aadbfc24f5 --- /dev/null +++ b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/statsd/StatsDCounterKey.java @@ -0,0 +1,8 @@ +package datadog.metrics.api.statsd; + +/** A counter identity: the dogstatsd metric name and tags a batch of counts should report under. */ +public interface StatsDCounterKey { + String getMetricName(); + + String[] getTags(); +} diff --git a/products/metrics/metrics-api/src/test/java/datadog/metrics/api/statsd/RecordingStatsDClient.java b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/statsd/RecordingStatsDClient.java new file mode 100644 index 00000000000..f017a134086 --- /dev/null +++ b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/statsd/RecordingStatsDClient.java @@ -0,0 +1,63 @@ +package datadog.metrics.api.statsd; + +import java.util.ArrayList; +import java.util.List; + +/** Test fake that records every {@link #count} call for assertion. */ +final class RecordingStatsDClient implements StatsDClient { + + static final class Count { + final String metricName; + final long delta; + final String[] tags; + + Count(String metricName, long delta, String[] tags) { + this.metricName = metricName; + this.delta = delta; + this.tags = tags; + } + } + + final List counts = new ArrayList<>(); + + @Override + public void incrementCounter(String metricName, String... tags) {} + + @Override + public void count(String metricName, long delta, String... tags) { + counts.add(new Count(metricName, delta, tags)); + } + + @Override + public void gauge(String metricName, long value, String... tags) {} + + @Override + public void gauge(String metricName, double value, String... tags) {} + + @Override + public void histogram(String metricName, long value, String... tags) {} + + @Override + public void histogram(String metricName, double value, String... tags) {} + + @Override + public void distribution(String metricName, long value, String... tags) {} + + @Override + public void distribution(String metricName, double value, String... tags) {} + + @Override + public void serviceCheck( + String serviceCheckName, String status, String message, String... tags) {} + + @Override + public void error(Exception error) {} + + @Override + public int getErrorCount() { + return 0; + } + + @Override + public void close() {} +} diff --git a/products/metrics/metrics-api/src/test/java/datadog/metrics/api/statsd/StatsDCountReporterTest.java b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/statsd/StatsDCountReporterTest.java new file mode 100644 index 00000000000..9324fa62aa3 --- /dev/null +++ b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/statsd/StatsDCountReporterTest.java @@ -0,0 +1,207 @@ +package datadog.metrics.api.statsd; + +import static org.junit.jupiter.api.Assertions.assertArrayEquals; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; + +import datadog.metrics.api.Accumulator; +import java.util.ArrayList; +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import java.util.concurrent.atomic.AtomicBoolean; +import org.junit.jupiter.api.Test; + +class StatsDCountReporterTest { + + private static final String[] TAG_A = {"env:a"}; + private static final String[] TAG_B = {"env:b"}; + + enum Counters implements StatsDCounterKey { + FOO("foo.total", TAG_A), + BAR("bar.total", TAG_A), + SHARED_A("shared.total", TAG_A), + SHARED_B("shared.total", TAG_B); + + private final String metricName; + private final String[] tags; + + Counters(String metricName, String[] tags) { + this.metricName = metricName; + this.tags = tags; + } + + @Override + public String getMetricName() { + return metricName; + } + + @Override + public String[] getTags() { + return tags; + } + } + + @Test + void reportsNonZeroCounterWithItsOwnMetricNameAndTags() { + RecordingStatsDClient statsD = new RecordingStatsDClient(); + Map deltas = new HashMap<>(); + deltas.put(Counters.FOO, 3L); + + StatsDCountReporter.report(statsD, Counters.values(), c -> deltas.getOrDefault(c, 0L)); + + assertEquals(1, statsD.counts.size()); + RecordingStatsDClient.Count count = statsD.counts.get(0); + assertEquals("foo.total", count.metricName); + assertEquals(3L, count.delta); + assertArrayEquals(TAG_A, count.tags); + } + + @Test + void skipsZeroDeltaCounters() { + RecordingStatsDClient statsD = new RecordingStatsDClient(); + + StatsDCountReporter.report(statsD, Counters.values(), c -> 0L); + + assertTrue(statsD.counts.isEmpty()); + } + + @Test + void reportsNothingWhenEveryCounterIsZero() { + RecordingStatsDClient statsD = new RecordingStatsDClient(); + Map deltas = new HashMap<>(); + + StatsDCountReporter.report(statsD, Counters.values(), c -> deltas.getOrDefault(c, 0L)); + + assertTrue(statsD.counts.isEmpty()); + } + + @Test + void reportsConstantsSharingAMetricNameIndependentlyByTag() { + RecordingStatsDClient statsD = new RecordingStatsDClient(); + Map deltas = new HashMap<>(); + deltas.put(Counters.SHARED_A, 5L); + deltas.put(Counters.SHARED_B, 7L); + + StatsDCountReporter.report(statsD, Counters.values(), c -> deltas.getOrDefault(c, 0L)); + + assertEquals(2, statsD.counts.size()); + RecordingStatsDClient.Count a = statsD.counts.get(0); + RecordingStatsDClient.Count b = statsD.counts.get(1); + assertEquals("shared.total", a.metricName); + assertEquals(5L, a.delta); + assertArrayEquals(TAG_A, a.tags); + assertEquals("shared.total", b.metricName); + assertEquals(7L, b.delta); + assertArrayEquals(TAG_B, b.tags); + } + + @Test + void flushDrainsAndReportsWithoutPerturbingTheCumulativeLiveTotal() { + RecordingStatsDClient statsD = new RecordingStatsDClient(); + StatsDCountReporter reporter = StatsDCountReporter.of(statsD, Counters.class); + reporter.inc(Counters.FOO); + reporter.add(Counters.BAR, 4L); + + reporter.flush(); + + assertEquals(2, statsD.counts.size()); + // live() is a cumulative total (real events observed), not "since the last flush" -- it + // doesn't reset just because a flush drained and reported the delta. + Accumulator.Counts live = reporter.live(); + assertEquals(1L, live.get(Counters.FOO)); + assertEquals(4L, live.get(Counters.BAR)); + + reporter.flush(); + + // a second flush with nothing new to report doesn't inflate the cumulative total either. + assertEquals(2, statsD.counts.size()); + live = reporter.live(); + assertEquals(1L, live.get(Counters.FOO)); + assertEquals(4L, live.get(Counters.BAR)); + } + + @Test + void flushCompensatesCountersNotConfirmedDeliveredAfterAnException() { + List delivered = new ArrayList<>(); + AtomicBoolean failNextBar = new AtomicBoolean(true); + StatsDClient failsOnBarOnce = + new StatsDClient() { + @Override + public void incrementCounter(String metricName, String... tags) {} + + @Override + public void count(String metricName, long delta, String... tags) { + if (metricName.equals("bar.total") && failNextBar.compareAndSet(true, false)) { + throw new RuntimeException("boom"); + } + delivered.add(new RecordingStatsDClient.Count(metricName, delta, tags)); + } + + @Override + public void gauge(String metricName, long value, String... tags) {} + + @Override + public void gauge(String metricName, double value, String... tags) {} + + @Override + public void histogram(String metricName, long value, String... tags) {} + + @Override + public void histogram(String metricName, double value, String... tags) {} + + @Override + public void distribution(String metricName, long value, String... tags) {} + + @Override + public void distribution(String metricName, double value, String... tags) {} + + @Override + public void serviceCheck( + String serviceCheckName, String status, String message, String... tags) {} + + @Override + public void error(Exception error) {} + + @Override + public int getErrorCount() { + return 0; + } + + @Override + public void close() {} + }; + + StatsDCountReporter reporter = StatsDCountReporter.of(failsOnBarOnce, Counters.class); + reporter.inc(Counters.FOO); + reporter.add(Counters.BAR, 4L); + reporter.add(Counters.SHARED_A, 2L); + + reporter.flush(); + + assertEquals(1, delivered.size()); + assertEquals("foo.total", delivered.get(0).metricName); + + // the cumulative live total counts every real event exactly once, whether or not statsd ever + // received it -- compensating for the failed send must not inflate this. + Accumulator.Counts live = reporter.live(); + assertEquals(1L, live.get(Counters.FOO)); + assertEquals(4L, live.get(Counters.BAR)); + assertEquals(2L, live.get(Counters.SHARED_A)); + + delivered.clear(); + reporter.flush(); + + assertEquals(2, delivered.size()); + assertEquals("bar.total", delivered.get(0).metricName); + assertEquals(4L, delivered.get(0).delta); + assertEquals("shared.total", delivered.get(1).metricName); + assertEquals(2L, delivered.get(1).delta); + + // still exactly-once in the cumulative total after the retry succeeds. + live = reporter.live(); + assertEquals(1L, live.get(Counters.FOO)); + assertEquals(4L, live.get(Counters.BAR)); + assertEquals(2L, live.get(Counters.SHARED_A)); + } +} From 5493087c6852a05883b1854f875d65ae4cf5a400 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Thu, 10 Sep 2026 18:14:51 -0400 Subject: [PATCH 30/36] Return an unmodifiable List view from Counts.keys() instead of the shared array Protects the backing key array every Counts from the same Accumulator shares, without paying for a defensive copy. Also clarifies RunningTotal.of()'s javadoc about its pre-existing-value seeding visibility gap. Co-Authored-By: Claude Sonnet 5 --- .../java/datadog/metrics/api/Accumulator.java | 18 ++++++++++++++---- .../api/statsd/StatsDCountReporter.java | 9 +++++---- .../datadog/metrics/api/AccumulatorTest.java | 10 +++++++++- 3 files changed, 28 insertions(+), 9 deletions(-) diff --git a/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java index 6a44390d405..2a5140adac8 100644 --- a/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java +++ b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java @@ -1,6 +1,9 @@ package datadog.metrics.api; import datadog.environment.ThreadSupport; +import java.util.Arrays; +import java.util.Collections; +import java.util.List; import java.util.concurrent.atomic.AtomicLongArray; import javax.annotation.concurrent.ThreadSafe; @@ -140,10 +143,13 @@ public long get(E key) { /** * The enum constants this {@link Counts} is keyed by, in declaration order -- for a caller that * wants to iterate every counter (e.g. reporting each one) without separately having to pass - * {@code E.values()} alongside this object. + * {@code E.values()} alongside this object. An unmodifiable view over the same backing array + * every {@link Counts} from the same {@link Accumulator} shares -- no copy, but a caller can't + * corrupt that shared array's ordering for every other snapshot the way a raw array reference + * would let it. */ - public E[] keys() { - return keys; + public List keys() { + return Collections.unmodifiableList(Arrays.asList(keys)); } /** @@ -199,7 +205,11 @@ private RunningTotal(Accumulator accumulator) { /** * Wraps {@code accumulator}, seeding the running total from a drain of whatever it currently - * holds (typically nothing, for a freshly constructed {@code accumulator}). + * holds. That seed becomes visible through {@link #live}, but -- being folded in at + * construction, before any caller can observe it -- it is never returned by a later {@link + * #drain}. A caller that reports each {@link #drain} delta (e.g. to statsd) would therefore + * silently never report pre-existing counts. Always construct this from a freshly created + * {@code accumulator} (as every current caller does) so there is nothing to seed. */ public static > RunningTotal of(Accumulator accumulator) { return new RunningTotal<>(accumulator); diff --git a/products/metrics/metrics-api/src/main/java/datadog/metrics/api/statsd/StatsDCountReporter.java b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/statsd/StatsDCountReporter.java index 0017039c80a..b149b422f4c 100644 --- a/products/metrics/metrics-api/src/main/java/datadog/metrics/api/statsd/StatsDCountReporter.java +++ b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/statsd/StatsDCountReporter.java @@ -1,6 +1,7 @@ package datadog.metrics.api.statsd; import datadog.metrics.api.Accumulator; +import java.util.List; import java.util.function.ToLongFunction; import javax.annotation.concurrent.ThreadSafe; import org.slf4j.Logger; @@ -85,9 +86,9 @@ public void flush() { * holding whatever wasn't attempted or confirmed sent */ private Accumulator.Counts report(Accumulator.Counts counts) { - E[] keys = counts.keys(); - for (int i = 0; i < keys.length; i++) { - E key = keys[i]; + List keys = counts.keys(); + for (int i = 0; i < keys.size(); i++) { + E key = keys.get(i); long delta = counts.get(key); if (delta != 0) { try { @@ -96,7 +97,7 @@ private Accumulator.Counts report(Accumulator.Counts counts) { log.debug( "Failed to report {}, compensating {} undelivered counter(s)", key, - keys.length - i, + keys.size() - i, e); return counts.from(i); } diff --git a/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java index 3d2592fd522..57e39a2f11f 100644 --- a/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java +++ b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java @@ -1,6 +1,7 @@ package datadog.metrics.api; import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertThrows; import static org.junit.jupiter.api.Assertions.assertTrue; import java.util.concurrent.CountDownLatch; @@ -199,7 +200,7 @@ void countsExposesItsOwnKeysWithoutASeparateValuesArray() { counters.add(Counters.BAR, 5L); Accumulator.Counts drained = counters.accumulateAndReset(); - assertEquals(Counters.values().length, drained.keys().length); + assertEquals(Counters.values().length, drained.keys().size()); long total = 0L; for (Counters c : drained.keys()) { @@ -208,6 +209,13 @@ void countsExposesItsOwnKeysWithoutASeparateValuesArray() { assertEquals(6L, total); } + @Test + void keysIsUnmodifiable() { + Accumulator counters = Accumulator.of(Counters.class); + Accumulator.Counts drained = counters.accumulateAndReset(); + assertThrows(UnsupportedOperationException.class, () -> drained.keys().set(0, Counters.BAZ)); + } + @Test void plusCombinesAStoredRunningTotalWithALiveSumWithoutMutatingEither() { Accumulator counters = Accumulator.of(Counters.class); From a60d97c6ce4382679556986ebb58ad6417a05e32 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Thu, 10 Sep 2026 18:15:07 -0400 Subject: [PATCH 31/36] Cap Accumulator's stripe count at 64 regardless of core count Without a cap, stripeCount() grows unbounded on very-high-core-count hosts. Past that many contending threads the collision-reduction math has already flattened out, so an unbounded table isn't worth the memory cost. Co-Authored-By: Claude Sonnet 5 --- .../main/java/datadog/metrics/api/Accumulator.java | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java index 2a5140adac8..28c55a59be0 100644 --- a/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java +++ b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java @@ -40,6 +40,9 @@ public final class Accumulator> { /** One full cache line of {@code long}s (64 bytes), used to pad each stripe row. */ private static final int CACHE_LINE_LONGS = 8; + /** Upper bound on {@link #stripeCount()}, regardless of core count. */ + private static final int MAX_STRIPES = 64; + private final AtomicLongArray[] data; private final int width; private final E[] keys; @@ -255,8 +258,9 @@ private static AtomicLongArray stripeOf(AtomicLongArray[] data) { /** * A fixed, power-of-two stripe count deliberately oversized to roughly 2x {@link - * Runtime#availableProcessors()} (minimum 4). Not exposed as a per-call override: a mandatory - * sizing knob on every caller fails the "print test" of self-explanatory API design. + * Runtime#availableProcessors()} (minimum 4, maximum {@link #MAX_STRIPES}). Not exposed as a + * per-call override: a mandatory sizing knob on every caller fails the "print test" of + * self-explanatory API design. * *

Sizing to exactly the core count leaves stripe collisions likely under real contention * (birthday-paradox math: with {@code n} contending threads and {@code m} stripes, expected @@ -266,10 +270,14 @@ private static AtomicLongArray stripeOf(AtomicLongArray[] data) { * cost, at the price of a slightly more expensive (but far rarer) {@link #accumulateAndReset} * drain -- the right trade given {@link #inc}/{@link #add} run on every call while {@link * #accumulateAndReset} runs on a reporting cadence. + * + *

Capped at {@link #MAX_STRIPES} so a single {@link Accumulator} on a very-high-core-count + * host can't grow its fixed table without bound: past that many contending threads, collision + * math has already flattened out, so the memory cost of chasing it further isn't worth paying. */ private static int stripeCount() { int cpus = Runtime.getRuntime().availableProcessors(); - return Math.max(4, 2 * Integer.highestOneBit(Math.max(1, cpus))); + return Math.min(MAX_STRIPES, Math.max(4, 2 * Integer.highestOneBit(Math.max(1, cpus)))); } /** Rounds {@code width} up to a whole number of cache lines, plus one full trailing line. */ From 2fe50b1550de551bf6c140e2c3d03f065ed12020 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Thu, 10 Sep 2026 20:45:50 -0400 Subject: [PATCH 32/36] Refresh AccumulatorBenchmark javadoc results after the stripe-count cap Re-ran the Fork(5) suite post-MAX_STRIPES=64 to confirm the write-side win and drain-side parity still hold; numbers moved slightly but the conclusions are unchanged. Co-Authored-By: Claude Sonnet 5 --- .../metrics/api/AccumulatorBenchmark.java | 46 +++++++++---------- 1 file changed, 23 insertions(+), 23 deletions(-) diff --git a/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorBenchmark.java b/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorBenchmark.java index d7c1779165f..a34d894ea63 100644 --- a/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorBenchmark.java +++ b/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorBenchmark.java @@ -51,34 +51,34 @@ * {@code longAdderGroup8}'s threads split into up to 8 groups each contending their own lock -- the * topology where distributed locking should actually pay off, forcing {@link Accumulator}'s * thread-striped design to earn its write-side win rather than facing a single-counter worst case. - * Fork(5), 15 samples per benchmark: - * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 15 2.746 ± 0.050 us/op - * AccumulatorBenchmark.accumulatorAccumulateAndReset_lowContention avgt 15 0.056 ± 0.001 us/op - * AccumulatorBenchmark.accumulatorAccumulateAndReset8_highContention avgt 15 6.875 ± 0.422 us/op - * AccumulatorBenchmark.accumulatorAccumulateAndReset8_lowContention avgt 15 0.363 ± 0.003 us/op + * Fork(5), 15 samples per benchmark, Apple M1 Max, 10 CPUs - macOS/aarch64 - JDK 25 (Zulu): + * AccumulatorBenchmark.accumulatorAccumulateAndReset_highContention avgt 15 2.760 ± 0.052 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset_lowContention avgt 15 0.049 ± 0.001 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset8_highContention avgt 15 6.939 ± 0.303 us/op + * AccumulatorBenchmark.accumulatorAccumulateAndReset8_lowContention avgt 15 0.364 ± 0.003 us/op * AccumulatorBenchmark.accumulatorIncrement_highContention avgt 15 0.009 ± 0.001 us/op * AccumulatorBenchmark.accumulatorIncrement_lowContention avgt 15 0.007 ± 0.001 us/op - * AccumulatorBenchmark.accumulatorIncrement8_highContention avgt 15 0.017 ± 0.009 us/op + * AccumulatorBenchmark.accumulatorIncrement8_highContention avgt 15 0.016 ± 0.006 us/op * AccumulatorBenchmark.accumulatorIncrement8_lowContention avgt 15 0.007 ± 0.001 us/op - * AccumulatorBenchmark.longAdderGroupAccumulateAnd_highContention avgt 15 4.770 ± 1.795 us/op - * AccumulatorBenchmark.longAdderGroupAccumulateAnd_lowContention avgt 15 0.061 ± 0.007 us/op - * AccumulatorBenchmark.longAdderGroupAccumulateAnd8_highContention avgt 15 6.025 ± 0.712 us/op - * AccumulatorBenchmark.longAdderGroupAccumulateAnd8_lowContention avgt 15 0.085 ± 0.004 us/op - * AccumulatorBenchmark.longAdderGroupIncrement_highContention avgt 15 2.294 ± 0.101 us/op - * AccumulatorBenchmark.longAdderGroupIncrement_lowContention avgt 15 0.019 ± 0.001 us/op - * AccumulatorBenchmark.longAdderGroupIncrement8_highContention avgt 15 0.786 ± 0.078 us/op - * AccumulatorBenchmark.longAdderGroupIncrement8_lowContention avgt 15 0.020 ± 0.001 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd_highContention avgt 15 1.703 ± 1.785 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd_lowContention avgt 15 0.024 ± 0.001 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd8_highContention avgt 15 5.989 ± 0.241 us/op + * AccumulatorBenchmark.longAdderGroupAccumulateAnd8_lowContention avgt 15 0.074 ± 0.008 us/op + * AccumulatorBenchmark.longAdderGroupIncrement_highContention avgt 15 2.775 ± 0.531 us/op + * AccumulatorBenchmark.longAdderGroupIncrement_lowContention avgt 15 0.012 ± 0.001 us/op + * AccumulatorBenchmark.longAdderGroupIncrement8_highContention avgt 15 0.513 ± 0.094 us/op + * AccumulatorBenchmark.longAdderGroupIncrement8_lowContention avgt 15 0.012 ± 0.001 us/op * On the write side, {@link Accumulator} beats {@code longAdderGroup} at high contention by - * ~255x in the degenerate single-shared-lock case and still by ~46x once counters are fairly spread + * ~310x in the degenerate single-shared-lock case and still by ~32x once counters are fairly spread * across 8 locks -- a large, reproducible win either way, on the call that runs on every event. On - * the drain side, the two designs are close and the comparison is noisy under contention for both: - * at width 1 {@link Accumulator}'s drain (2.746 us/op) is actually faster than {@code - * longAdderGroup}'s (4.770 ± 1.795 us/op, itself high-variance), and at width 8 it's only ~1.14x - * slower (6.875 vs 6.025 us/op) -- not the regression an earlier reading of this benchmark - * suggested. That earlier reading (13.357 us/op at Fork(2)) turned out to be a correlated anomaly - * across two independent low-sample runs, not a reproducible result; escalating to Fork(5) (15 - * samples) settled it. Net: a large, robust win on the call that fires on every event, and no - * confirmed cost on the call that fires once per reporting cycle. + * the drain side, the two designs remain close and the comparison stays noisy under contention for + * both: at width 1 {@link Accumulator}'s drain (2.760 us/op) reads slower than {@code + * longAdderGroup}'s (1.703 ± 1.785 us/op), but that error bar spans {@link Accumulator}'s own + * result, so the two aren't distinguishable at this sample size; at width 8 it's ~1.16x slower + * (6.939 vs 5.989 us/op), consistent with the earlier ~1.14x reading. Reproduced on a second, + * independent Fork(5) run after the stripe-count cap ({@code MAX_STRIPES = 64}) landed: a large, + * robust win on the call that fires on every event, and no confirmed cost on the call that fires + * once per reporting cycle. */ @State(Scope.Benchmark) @Warmup(iterations = 1, time = 10) From 300ca8ddd76dd607ca448674875f49d9254c891f Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Thu, 10 Sep 2026 21:35:53 -0400 Subject: [PATCH 33/36] Run AccumulatorVsCounterBenchmark with a real client, not a true no-op StatsDClient.NO_OP does nothing, so counterIncrement was measuring bare virtual-dispatch cost instead of StatsDCounter's real per-call cost -- the comparison couldn't answer its own question, and made Counter look cheaper than Accumulator even at high contention. Swap in a LockingStatsDClient that pays a real shared-lock cost per call, mirroring every real StatsDClient's single shared connection. Document the corrected results in the class javadoc. Co-Authored-By: Claude Sonnet 5 --- .../api/AccumulatorVsCounterBenchmark.java | 93 +++++++++++++++++-- 1 file changed, 85 insertions(+), 8 deletions(-) diff --git a/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorVsCounterBenchmark.java b/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorVsCounterBenchmark.java index d0b600871d7..9ad0d622c62 100644 --- a/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorVsCounterBenchmark.java +++ b/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorVsCounterBenchmark.java @@ -25,13 +25,17 @@ *

{@link Counter} is the one most callers reach for first, and for good reason -- it's the * advertised, general-purpose metrics API. But its real implementation ({@code StatsDCounter} in * {@code metrics-lib}) calls {@link StatsDClient#count} synchronously on every {@link - * Counter#increment}, with no batching or striping of its own. {@link #counterIncrement} below - * mirrors that shape exactly (forwarding straight to a no-op {@link StatsDClient} to isolate the - * call-site cost from real network I/O, which would swamp everything else and isn't the question - * this asks -- {@code StatsDCounter} itself has package-private construction, so this reimplements - * its shape rather than depending on {@code metrics-lib}). {@link Accumulator} exists because that - * per-call cost is too high to pay on every request/span/event -- it stripes by thread and defers - * reporting to a periodic drain instead. + * Counter#increment}, with no batching or striping of its own, and the real {@code StatsDClient} + * underneath serializes that call through one shared connection (a lock or an offer to a single + * queue) -- the same "one shared serialization point, no thread distribution" shape {@link + * AccumulatorBenchmark}'s {@code longAdderGroup} single-lock variants model. {@link + * #counterIncrement} below mirrors that shape with a {@link StatsDClient} that takes a real lock on + * every call (see {@link LockingStatsDClient}) rather than a true no-op -- a true no-op measures + * only virtual-dispatch overhead and understates {@code StatsDCounter}'s actual per-call cost to + * the point of not answering this benchmark's own question ({@code StatsDCounter} itself has + * package-private construction, so this reimplements its shape rather than depending on {@code + * metrics-lib}). {@link Accumulator} exists because that per-call cost is too high to pay on every + * request/span/event -- it stripes by thread and defers reporting to a periodic drain instead. * *

Rule of thumb: a counter incremented on a hot path (every request, span, or event) * should use {@link Accumulator}, drained on a reporting cadence. A counter incremented rarely @@ -39,6 +43,20 @@ * directly, no ceremony required. This is the concrete case behind the (not yet built) * {@code @ForegroundSafe}/{@code @BackgroundOnly} annotations: {@link Counter#increment} is * {@code @BackgroundOnly}, {@link Accumulator#inc} is {@code @ForegroundSafe}. + * + *

Fork(5), 15 samples per benchmark, Apple M1 Max, 10 CPUs - macOS/aarch64 - JDK 25 (Zulu): + * + * AccumulatorVsCounterBenchmark.accumulatorIncrement_highContention avgt 15 0.009 ± 0.001 us/op + * AccumulatorVsCounterBenchmark.accumulatorIncrement_lowContention avgt 15 0.007 ± 0.001 us/op + * AccumulatorVsCounterBenchmark.counterIncrement_highContention avgt 15 1.559 ± 0.941 us/op + * AccumulatorVsCounterBenchmark.counterIncrement_lowContention avgt 15 0.009 ± 0.001 us/op + * At low contention the two are indistinguishable (0.007 vs 0.009 us/op) -- an uncontended + * lock costs almost nothing, so with only one thread ever calling in, {@link Counter}'s per-call + * cost and {@link Accumulator}'s are both dominated by the same handful of instructions. At high + * contention {@link Accumulator} wins by ~173x (0.009 vs 1.559 us/op, itself high-variance from run + * to run) -- {@link Counter}'s one shared lock serializes every calling thread, while {@link + * Accumulator}'s per-thread striping doesn't. This is the expected result for a counter hit from + * many concurrent threads, and it's why the Rule of thumb above exists. */ @State(Scope.Benchmark) @Warmup(iterations = 1, time = 10) @@ -53,7 +71,7 @@ enum Metric { } private final Accumulator accumulator = Accumulator.of(Metric.class); - private final Counter counter = new SynchronousStatsDCounter("hits", StatsDClient.NO_OP); + private final Counter counter = new SynchronousStatsDCounter("hits", new LockingStatsDClient()); /** * Mirrors {@code StatsDCounter}'s real shape: every {@link #increment} forwards straight to the @@ -81,6 +99,65 @@ public void incrementErrorCount(String cause, int delta) { } } + /** + * A {@link StatsDClient} stand-in that pays a real, serialized per-call cost instead of a true + * no-op: every real {@code StatsDClient} forwards {@code count} through one shared connection (a + * lock or an offer to a single non-blocking queue), so a true no-op would measure only + * virtual-dispatch overhead and understate the cost this benchmark exists to isolate. A {@code + * synchronized} increment of a shared counter is a reasonable stand-in for that shared + * serialization point without pulling in a real socket/queue implementation this benchmark + * doesn't need. + */ + private static final class LockingStatsDClient implements StatsDClient { + private final Object lock = new Object(); + private long total; + + @Override + public void incrementCounter(String metricName, String... tags) { + count(metricName, 1L, tags); + } + + @Override + public void count(String metricName, long delta, String... tags) { + synchronized (lock) { + total += delta; + } + } + + @Override + public void gauge(String metricName, long value, String... tags) {} + + @Override + public void gauge(String metricName, double value, String... tags) {} + + @Override + public void histogram(String metricName, long value, String... tags) {} + + @Override + public void histogram(String metricName, double value, String... tags) {} + + @Override + public void distribution(String metricName, long value, String... tags) {} + + @Override + public void distribution(String metricName, double value, String... tags) {} + + @Override + public void serviceCheck( + String serviceCheckName, String status, String message, String... tags) {} + + @Override + public void error(Exception error) {} + + @Override + public int getErrorCount() { + return 0; + } + + @Override + public void close() {} + } + @Benchmark @Threads(1) public void accumulatorIncrement_lowContention() { From aae3a54dfe881d42b7fb681ad819e2a2dfd27e35 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Thu, 10 Sep 2026 21:47:00 -0400 Subject: [PATCH 34/36] Rename Counts.zero() to Counts.create() Co-Authored-By: Claude Sonnet 5 --- .../src/main/java/datadog/metrics/api/Accumulator.java | 2 +- .../src/test/java/datadog/metrics/api/AccumulatorTest.java | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java index 28c55a59be0..4f2c618534e 100644 --- a/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java +++ b/products/metrics/metrics-api/src/main/java/datadog/metrics/api/Accumulator.java @@ -133,7 +133,7 @@ private Counts(long[] counts, E[] keys) { * * @param enumType the enum naming each counter, e.g. {@code MyCounters.class} */ - public static > Counts zero(Class enumType) { + public static > Counts create(Class enumType) { E[] keys = enumType.getEnumConstants(); return new Counts<>(new long[keys.length], keys); } diff --git a/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java index 57e39a2f11f..adb203fdd1c 100644 --- a/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java +++ b/products/metrics/metrics-api/src/test/java/datadog/metrics/api/AccumulatorTest.java @@ -181,8 +181,8 @@ void sumReflectsIncrementsMadeAfterAnEarlierSum() { } @Test - void zeroSeedsAnAllZeroCountsWithoutAScratchAccumulator() { - Accumulator.Counts zero = Accumulator.Counts.zero(Counters.class); + void createSeedsAnAllZeroCountsWithoutAScratchAccumulator() { + Accumulator.Counts zero = Accumulator.Counts.create(Counters.class); assertEquals(0L, zero.get(Counters.FOO)); assertEquals(0L, zero.get(Counters.BAR)); From 9f4cc1b601ccef613c5618cb8ec76401058d2ff4 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Fri, 11 Sep 2026 06:52:31 -0400 Subject: [PATCH 35/36] Add fair single-counter LongAdder-delta baseline to AccumulatorBenchmark sumThenReset() isn't the only safe alternative to Accumulator for one counter: a single dedicated differ computing sum() - previous by hand (the pattern pre-migration TracerHealthMetrics actually used) is also lock-free and never loses an update. Add that baseline (longAdderDelta*/ longAdderDeltaMixed*) and rewrite the class javadoc to reframe the existing longAdderGroup* win as the cost of a lock-based guarantee, not of LongAdder itself, and to note that Accumulator's real, decisive evidence is the batch/drain win demonstrated on real code in TracerHealthMetricsBenchmark rather than these single-counter numbers. Co-Authored-By: Claude Sonnet 5 --- .../metrics/api/AccumulatorBenchmark.java | 98 +++++++++++++++++++ 1 file changed, 98 insertions(+) diff --git a/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorBenchmark.java b/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorBenchmark.java index a34d894ea63..f5ab5147c1c 100644 --- a/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorBenchmark.java +++ b/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorBenchmark.java @@ -79,6 +79,51 @@ * independent Fork(5) run after the stripe-count cap ({@code MAX_STRIPES = 64}) landed: a large, * robust win on the call that fires on every event, and no confirmed cost on the call that fires * once per reporting cycle. + * + *

{@code longAdderDelta*}: the fair single-counter baseline. {@code longAdderGroup}'s + * ~310x/~32x win above is real, but it's the cost of buying {@code sumThenReset()}'s no-lost-update + * guarantee via a lock -- not the cost of a single {@code LongAdder} on its own. + * Pre-migration {@code TracerHealthMetrics} never took that lock: it kept a single {@code + * previousCounts}/{@code countIndex}-tracked differ computing {@code sum() - previous} by hand, + * which is already lock-free and never loses an update, because a missed delta on one {@code sum()} + * just shows up whole on the next one (see {@link #deltaSumAndReset}). {@code longAdderDelta_*}/ + * {@code longAdderDeltaMixed_*} below reproduce exactly that pattern -- one {@code LongAdder}, one + * dedicated differ, no lock anywhere -- as the fairest single-counter comparison to {@link + * Accumulator}: both designs give the same no-lost-update guarantee, just by different means + * (per-thread striping vs. a single differ's own unsynchronized bookkeeping), so neither pays for a + * lock the other doesn't need. Fork(5), 15 samples per benchmark, same machine: + * AccumulatorBenchmark.accumulatorAccumulateAndReset_lowContention avgt 15 0.050 ± 0.001 us/op + * AccumulatorBenchmark.accumulatorMixed avgt 15 0.158 ± 0.022 us/op + * AccumulatorBenchmark.accumulatorMixed:accumulatorMixed_drain avgt 15 0.692 ± 0.077 us/op + * AccumulatorBenchmark.accumulatorMixed:accumulatorMixed_write avgt 15 0.024 ± 0.009 us/op + * AccumulatorBenchmark.longAdderDelta_lowContention avgt 15 0.018 ± 0.001 us/op + * AccumulatorBenchmark.longAdderDeltaMixed avgt 15 0.118 ± 0.006 us/op + * AccumulatorBenchmark.longAdderDeltaMixed:longAdderDeltaMixed_drain avgt 15 0.122 ± 0.012 us/op + * AccumulatorBenchmark.longAdderDeltaMixed:longAdderDeltaMixed_write avgt 15 0.117 ± 0.006 us/op + * Here {@link Accumulator} does not win outright. In the single-threaded + * inc-then-diff-per-call shape, the safe {@code LongAdder} delta is ~2.8x cheaper (0.018 vs 0.050 + * us/op) -- no striping to fan out or fold back in when there's only one thread. In the realistic + * many-writers/one-drainer topology ({@code accumulatorMixed} vs {@code longAdderDeltaMixed}), + * {@link Accumulator} wins the write side by ~4.9x (0.024 vs 0.117 us/op, the call on the hot path) + * but loses the drain side by ~5.7x (0.692 vs 0.122 us/op) and the combined total by ~1.34x (0.158 + * vs 0.118 us/op) -- the striping that makes writes cheap has to be folded back together somewhere, + * and that fold costs more than one differ's plain subtraction. + * + *

That's not a mark against {@link Accumulator}: it's the expected shape of a primitive whose + * value isn't raw single-counter throughput. {@link Accumulator}'s real win shows up one level up, + * in a from-scratch before/after of an actual migrated caller -- {@code + * TracerHealthMetricsBenchmark} (see {@code + * dd-trace-core/src/jmh/java/datadog/trace/core/monitor/}), which replaced ~49 individual {@code + * LongAdder} fields and their hand-rolled {@code previousCounts}/{@code countIndex} delta tracking + * with one {@code Accumulator}. There, every hot single-counter call is at + * parity with or faster than the legacy code (0.6-0.7x of legacy on {@code onSend}, the most + * frequent real call site), and the batch drain -- the actual shape {@link Accumulator} is for, + * many counters read and reset together once per reporting cycle -- is where the real-code evidence + * is decisive, not just competitive. The fair single-counter numbers above are worth publishing + * precisely because they're not a clean win: they show {@link Accumulator} doesn't need to dominate + * every synthetic one-counter microbenchmark to be the right call once counters are plural and the + * drain is the thing that matters, which {@code TracerHealthMetricsBenchmark} demonstrates directly + * on real code rather than a synthetic stand-in. */ @State(Scope.Benchmark) @Warmup(iterations = 1, time = 10) @@ -120,6 +165,8 @@ enum Counter8 { private static final Counter8[] COUNTER8_VALUES = Counter8.values(); private final LongAdder adder = new LongAdder(); + private final LongAdder deltaAdder = new LongAdder(); + private long deltaPrevious; private final Accumulator accumulator = Accumulator.of(Counter.class); private final Accumulator accumulator8 = Accumulator.of(Counter8.class); private final ConcurrentHashMap chm = new ConcurrentHashMap<>(); @@ -191,6 +238,29 @@ private static long[] arrayAccumulateAndReset(AtomicLongArray array) { return acc; } + /** + * The safe, lock-free alternative to {@code sumThenReset()} that pre-migration {@code + * TracerHealthMetrics} actually used (via {@code previousCounts}/{@code countIndex}): never reset + * the {@code LongAdder} at all, and have a single differ thread track the last observed {@code + * sum()} to compute its own delta. {@code sum()} alone never loses an update permanently -- a + * miss just shows up in the next {@code sum()} -- so this closes {@code sumThenReset()}'s reset + * race without any lock, at the cost of one extra subtraction per drain. The catch is the "single + * differ" part: {@code deltaPrevious} is unsynchronized plain state, correct only because exactly + * one thread ever calls this method between increments. Unlike {@code sumThenReset()}, which + * degrades gracefully (just an occasional dropped delta) if called from multiple threads at once, + * concurrent callers here would race on {@code deltaPrevious} itself and corrupt it -- so there + * is deliberately no {@code longAdderDelta_highContention} mirroring {@code + * longAdderSumThenReset_highContention}'s "every thread both writes and drains" shape; see {@code + * longAdderDeltaMixed_write}/{@code _drain} below for the one topology (many writers, one + * dedicated drainer) this baseline is actually valid under. + */ + private long deltaSumAndReset() { + long current = deltaAdder.sum(); + long delta = current - deltaPrevious; + deltaPrevious = current; + return delta; + } + @Benchmark @Threads(1) public void longAdderIncrement_lowContention() { @@ -241,6 +311,34 @@ public void longAdderSumThenReset_highContention(Blackhole blackhole) { blackhole.consume(adder.sumThenReset()); } + @Benchmark + @Threads(1) + public void longAdderDelta_lowContention(Blackhole blackhole) { + deltaAdder.increment(); + blackhole.consume(deltaSumAndReset()); + } + + /** + * The realistic, valid topology for {@link #deltaSumAndReset} -- many writers, one dedicated + * differ -- mirroring {@code accumulatorMixed_write}/{@code _drain} below so the two can be + * compared directly: this is the fairest single-counter match for {@link Accumulator}, since both + * are lock-free on the write side and both give the same no-lost-update guarantee, just by + * different means (per-slot atomic {@code getAndSet} vs. a single differ's own bookkeeping). + */ + @Benchmark + @Group("longAdderDeltaMixed") + @GroupThreads(4) + public void longAdderDeltaMixed_write() { + deltaAdder.increment(); + } + + @Benchmark + @Group("longAdderDeltaMixed") + @GroupThreads(1) + public void longAdderDeltaMixed_drain(Blackhole blackhole) { + blackhole.consume(deltaSumAndReset()); + } + @Benchmark @Threads(1) public void accumulatorAccumulateAndReset_lowContention(Blackhole blackhole) { From e9cc2de32350ed08646e1258631bf5ec29649622 Mon Sep 17 00:00:00 2001 From: Douglas Q Hawkins Date: Fri, 11 Sep 2026 14:03:04 -0400 Subject: [PATCH 36/36] Note AccumulatorVsCounterBenchmark's LockingStatsDClient as a conservative lower bound LockingStatsDClient stands in for a real StatsDClient's shared lock/queue but skips the encoding and queue-offer/socket I/O work that lock also guards in production, so the measured gap is a floor on the real one, not an exact measurement of it. Co-Authored-By: Claude Sonnet 5 --- .../metrics/api/AccumulatorVsCounterBenchmark.java | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorVsCounterBenchmark.java b/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorVsCounterBenchmark.java index 9ad0d622c62..46306a4fffc 100644 --- a/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorVsCounterBenchmark.java +++ b/products/metrics/metrics-api/src/jmh/java/datadog/metrics/api/AccumulatorVsCounterBenchmark.java @@ -57,6 +57,14 @@ * to run) -- {@link Counter}'s one shared lock serializes every calling thread, while {@link * Accumulator}'s per-thread striping doesn't. This is the expected result for a counter hit from * many concurrent threads, and it's why the Rule of thumb above exists. + * + *

What this does and doesn't measure. {@link LockingStatsDClient} is a conservative + * synthetic lower bound on {@link Counter}'s real cost, not a measurement of exact production + * overhead -- a real {@code StatsDClient} also encodes the metric line and offers it to a queue (or + * blocks on socket I/O) under that same lock/queue, work this stand-in skips entirely. That means + * the true gap between {@link Counter} and {@link Accumulator} at high contention in production is + * at least as large as the ~173x shown here, quite possibly larger; this benchmark establishes a + * floor, not a ceiling, on the win. */ @State(Scope.Benchmark) @Warmup(iterations = 1, time = 10)