Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
38 commits
Select commit Hold shift + click to select a range
c3def00
Add Accumulator: a striped long counter primitive as an alternative t…
dougqh Aug 31, 2026
48e01bb
Oversize Accumulator's default stripe count to reduce contention coll…
dougqh Aug 31, 2026
3629c97
Add JOL footprint test: Accumulator vs one LongAdder per counter
dougqh Aug 31, 2026
fc2f55c
Benchmark Accumulator against a per-counter-locked LongAdder alternative
dougqh Aug 31, 2026
89d980c
Address review comments on Accumulator
dougqh Aug 31, 2026
7107b5b
Split Accumulator into a typed wrapper and a nested EmbeddingSupport
dougqh Sep 1, 2026
175154e
Update Accumulator tests/benchmark for the EmbeddingSupport split
dougqh Sep 1, 2026
c9f7fc2
Wrap Accumulator's update/accumulateAndReset in typed Stripe/Counts v…
dougqh Sep 1, 2026
a8e7448
Add benchmarks for Accumulator's typed API alongside raw EmbeddingSup…
dougqh Sep 1, 2026
bf67e0b
Record typed-vs-raw Accumulator benchmark results in AccumulatorBench…
dougqh Sep 1, 2026
3484ed7
Rename accumulatorAccumulateAnd* benchmarks, cap footprint test threa…
dougqh Sep 2, 2026
28de303
Add a non-destructive Accumulator.sum() for live diagnostic reads
dougqh Sep 2, 2026
3c5b7ef
Add Accumulator.Counts.plus() to combine a stored total with a live sum
dougqh Sep 2, 2026
93e2c2d
Add Accumulator.Counts.zero() to seed a running total without a scrat…
dougqh Sep 2, 2026
34f1830
Add a contextual Accumulator.update(context, BiConsumer) overload
dougqh Sep 2, 2026
8df6143
Let Counts expose its own keys and add Class<E>-based factories
dougqh Sep 2, 2026
9bc7abc
Rename Counts.values() to Counts.keys()
dougqh Sep 2, 2026
7e80a21
Add Accumulator.update(long, ObjLongConsumer) to avoid boxing a primi…
dougqh Sep 2, 2026
69251a5
Only touch the real counter width in combine/reset, not the padded st…
dougqh Sep 2, 2026
baf0a33
Replace Accumulator's synchronized stripes with lock-free AtomicLongA…
dougqh Sep 2, 2026
e90422c
Correct AccumulatorBenchmark javadoc's high-contention drain numbers
dougqh Sep 2, 2026
065ee24
Add fair per-thread-distributed 8-wide AccumulatorBenchmark compariso…
dougqh Sep 3, 2026
b90e7f6
Remove Accumulator.of(E[])/Counts.zero(E[]) in favor of the Class<E> …
dougqh Sep 10, 2026
97274e5
Add unstriped AtomicLongArray baseline to AccumulatorBenchmark
dougqh Sep 10, 2026
f28a67a
Add Accumulator.RunningTotal, an opt-in drain+live-read composition
dougqh Sep 10, 2026
631edde
Rename Accumulator/Counts field values -> keys
dougqh Sep 10, 2026
8e6e624
Move Accumulator from internal-api into metrics-api, next to Counter
dougqh Sep 10, 2026
b3075db
Mark Accumulator/RunningTotal @ThreadSafe; add Counts.from(fromIndex)
dougqh Sep 10, 2026
108ce69
Add StatsDCountReporter: an Accumulator-backed periodic statsd report…
dougqh Sep 10, 2026
5493087
Return an unmodifiable List view from Counts.keys() instead of the sh…
dougqh Sep 10, 2026
a60d97c
Cap Accumulator's stripe count at 64 regardless of core count
dougqh Sep 10, 2026
2fe50b1
Refresh AccumulatorBenchmark javadoc results after the stripe-count cap
dougqh Sep 11, 2026
300ca8d
Run AccumulatorVsCounterBenchmark with a real client, not a true no-op
dougqh Sep 11, 2026
aae3a54
Rename Counts.zero() to Counts.create()
dougqh Sep 11, 2026
9f4cc1b
Add fair single-counter LongAdder-delta baseline to AccumulatorBenchmark
dougqh Sep 11, 2026
e9cc2de
Note AccumulatorVsCounterBenchmark's LockingStatsDClient as a conserv…
dougqh Sep 11, 2026
48704bb
Merge branch 'master' into dougqh/accumulator-primitive
dougqh Sep 15, 2026
4be76af
Merge remote-tracking branch 'origin/master' into dougqh/accumulator-…
dougqh Sep 21, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions products/metrics/metrics-api/build.gradle.kts
Original file line number Diff line number Diff line change
@@ -1,15 +1,23 @@
plugins {
`java-library`
id("dd-trace-java.module.internal-api")
id("dd-trace-java.jmh-conventions")
}

description = "Metrics API"

dependencies {
implementation(libs.slf4j)
api(project(":components:environment"))

testImplementation(libs.bundles.junit5)
testImplementation(libs.bundles.mockito)
testImplementation(libs.jol.core)
}

jmh {
jmhVersion = libs.versions.jmh.get()
duplicateClassesStrategy = DuplicatesStrategy.EXCLUDE
}

extra["excludedClassesCoverage"] = listOf(
Expand Down

Large diffs are not rendered by default.

Original file line number Diff line number Diff line change
@@ -0,0 +1,192 @@
package datadog.metrics.api;

import static java.util.concurrent.TimeUnit.MICROSECONDS;

import datadog.metrics.api.statsd.StatsDClient;
import org.openjdk.jmh.annotations.Benchmark;
import org.openjdk.jmh.annotations.BenchmarkMode;
import org.openjdk.jmh.annotations.Fork;
import org.openjdk.jmh.annotations.Measurement;
import org.openjdk.jmh.annotations.Mode;
import org.openjdk.jmh.annotations.OutputTimeUnit;
import org.openjdk.jmh.annotations.Scope;
import org.openjdk.jmh.annotations.State;
import org.openjdk.jmh.annotations.Threads;
import org.openjdk.jmh.annotations.Warmup;

/**
* A decision-support benchmark, not a design-proof one: {@link AccumulatorBenchmark} exists to
* justify {@link Accumulator}'s internal striping design against synthetic alternatives ({@code
* LongAdder}, a CHM of {@code AtomicLong}, an unstriped {@code AtomicLongArray}) that nobody in
* this codebase would actually reach for instead -- that question is settled and doesn't need
* re-litigating on every read. This class answers a different, durable one: given both {@link
* Counter} and {@link Accumulator} live in {@code metrics-api}, which do you actually use?
*
* <p>{@link Counter} is the one most callers reach for first, and for good reason -- it's the
* advertised, general-purpose metrics API. But its real implementation ({@code StatsDCounter} in
* {@code metrics-lib}) calls {@link StatsDClient#count} synchronously on every {@link
* Counter#increment}, with no batching or striping of its own, and the real {@code StatsDClient}
* underneath serializes that call through one shared connection (a lock or an offer to a single
* queue) -- the same "one shared serialization point, no thread distribution" shape {@link
* AccumulatorBenchmark}'s {@code longAdderGroup} single-lock variants model. {@link
* #counterIncrement} below mirrors that shape with a {@link StatsDClient} that takes a real lock on
* every call (see {@link LockingStatsDClient}) rather than a true no-op -- a true no-op measures
* only virtual-dispatch overhead and understates {@code StatsDCounter}'s actual per-call cost to
* the point of not answering this benchmark's own question ({@code StatsDCounter} itself has
* package-private construction, so this reimplements its shape rather than depending on {@code
* metrics-lib}). {@link Accumulator} exists because that per-call cost is too high to pay on every
* request/span/event -- it stripes by thread and defers reporting to a periodic drain instead.
*
* <p><b>Rule of thumb:</b> a counter incremented on a hot path (every request, span, or event)
* should use {@link Accumulator}, drained on a reporting cadence. A counter incremented rarely
* (startup, config changes, an error path already off the hot path) can use {@link Counter}
* directly, no ceremony required. This is the concrete case behind the (not yet built)
* {@code @ForegroundSafe}/{@code @BackgroundOnly} annotations: {@link Counter#increment} is
* {@code @BackgroundOnly}, {@link Accumulator#inc} is {@code @ForegroundSafe}.
*
* <p>Fork(5), 15 samples per benchmark, Apple M1 Max, 10 CPUs - macOS/aarch64 - JDK 25 (Zulu):
* <code>
* AccumulatorVsCounterBenchmark.accumulatorIncrement_highContention avgt 15 0.009 ± 0.001 us/op
* AccumulatorVsCounterBenchmark.accumulatorIncrement_lowContention avgt 15 0.007 ± 0.001 us/op
* AccumulatorVsCounterBenchmark.counterIncrement_highContention avgt 15 1.559 ± 0.941 us/op
* AccumulatorVsCounterBenchmark.counterIncrement_lowContention avgt 15 0.009 ± 0.001 us/op
* </code> At low contention the two are indistinguishable (0.007 vs 0.009 us/op) -- an uncontended
* lock costs almost nothing, so with only one thread ever calling in, {@link Counter}'s per-call
* cost and {@link Accumulator}'s are both dominated by the same handful of instructions. At high
* contention {@link Accumulator} wins by ~173x (0.009 vs 1.559 us/op, itself high-variance from run
* to run) -- {@link Counter}'s one shared lock serializes every calling thread, while {@link
* Accumulator}'s per-thread striping doesn't. This is the expected result for a counter hit from
* many concurrent threads, and it's why the Rule of thumb above exists.
*
* <p><b>What this does and doesn't measure.</b> {@link LockingStatsDClient} is a conservative
* synthetic lower bound on {@link Counter}'s real cost, not a measurement of exact production
* overhead -- a real {@code StatsDClient} also encodes the metric line and offers it to a queue (or
* blocks on socket I/O) under that same lock/queue, work this stand-in skips entirely. That means
* the true gap between {@link Counter} and {@link Accumulator} at high contention in production is
* at least as large as the ~173x shown here, quite possibly larger; this benchmark establishes a
* floor, not a ceiling, on the win.
*/
@State(Scope.Benchmark)
@Warmup(iterations = 1, time = 10)
@Measurement(iterations = 3, time = 10)
@BenchmarkMode(Mode.AverageTime)
@OutputTimeUnit(MICROSECONDS)
@Fork(5)
public class AccumulatorVsCounterBenchmark {

enum Metric {
HITS
}

private final Accumulator<Metric> accumulator = Accumulator.of(Metric.class);
private final Counter counter = new SynchronousStatsDCounter("hits", new LockingStatsDClient());

/**
* Mirrors {@code StatsDCounter}'s real shape: every {@link #increment} forwards straight to the
* client, with no batching of its own. Reimplemented here rather than depending on {@code
* metrics-lib} because {@code StatsDCounter}'s constructor is package-private.
*/
private static final class SynchronousStatsDCounter implements Counter {
private static final String[] NO_TAGS = new String[0];
private final String name;
private final StatsDClient statsd;

SynchronousStatsDCounter(String name, StatsDClient statsd) {
this.name = name;
this.statsd = statsd;
}

@Override
public void increment(int delta) {
statsd.count(name, delta, NO_TAGS);
}

@Override
public void incrementErrorCount(String cause, int delta) {
statsd.count(name, delta, new String[] {"cause:" + cause});
}
}

/**
* A {@link StatsDClient} stand-in that pays a real, serialized per-call cost instead of a true
* no-op: every real {@code StatsDClient} forwards {@code count} through one shared connection (a
* lock or an offer to a single non-blocking queue), so a true no-op would measure only
* virtual-dispatch overhead and understate the cost this benchmark exists to isolate. A {@code
* synchronized} increment of a shared counter is a reasonable stand-in for that shared
* serialization point without pulling in a real socket/queue implementation this benchmark
* doesn't need.
*/
private static final class LockingStatsDClient implements StatsDClient {
private final Object lock = new Object();
private long total;

@Override
public void incrementCounter(String metricName, String... tags) {
count(metricName, 1L, tags);
}

@Override
public void count(String metricName, long delta, String... tags) {
synchronized (lock) {
total += delta;
}
}

@Override
public void gauge(String metricName, long value, String... tags) {}

@Override
public void gauge(String metricName, double value, String... tags) {}

@Override
public void histogram(String metricName, long value, String... tags) {}

@Override
public void histogram(String metricName, double value, String... tags) {}

@Override
public void distribution(String metricName, long value, String... tags) {}

@Override
public void distribution(String metricName, double value, String... tags) {}

@Override
public void serviceCheck(
String serviceCheckName, String status, String message, String... tags) {}

@Override
public void error(Exception error) {}

@Override
public int getErrorCount() {
return 0;
}

@Override
public void close() {}
}

@Benchmark
@Threads(1)
public void accumulatorIncrement_lowContention() {
accumulator.inc(Metric.HITS);
}

@Benchmark
@Threads(Threads.MAX)
public void accumulatorIncrement_highContention() {
accumulator.inc(Metric.HITS);
}

@Benchmark
@Threads(1)
public void counterIncrement_lowContention() {
counter.increment(1);
}

@Benchmark
@Threads(Threads.MAX)
public void counterIncrement_highContention() {
counter.increment(1);
}
}
Loading
Loading